Files
easyai-ai-gateway/scripts/cluster/run-production-acceptance.sh
T
wangbo 1bb4b2ab12 fix(acceptance): 同步配置 API 媒体并发
P24/P28/P32 之前只调整 Worker 媒体并发,API 仍使用默认每实例 16,导致入口实际容量低于 Worker 执行槽。现在按档位同时设置两地 API 的媒体请求与物化并发。\n\n验证:bash -n、ShellCheck。
2026-07-31 04:40:59 +08:00

1590 lines
62 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
script_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
# shellcheck source=scripts/cluster/common.sh
source "$script_dir/common.sh"
usage() {
cat <<'EOF'
Usage:
scripts/cluster/run-production-acceptance.sh --execute dist/releases/<SHA>.json
This is a destructive production acceptance workflow. It pauses new live tasks,
changes Worker capacity, creates paid real canaries, and deletes one Worker Pod
during each recovery phase. It also grants the dedicated acceptance API Key
access to the selected production candidates. A failed run deliberately keeps
traffic in validation mode and restores only the last stable capacity profile.
Required private environment values:
AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN
AI_GATEWAY_ACCEPTANCE_API_KEY
AI_GATEWAY_ACCEPTANCE_API_KEY_ID
AI_GATEWAY_ACCEPTANCE_USER_ID
AI_GATEWAY_ACCEPTANCE_GATEWAYS
AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS
Optional model overrides (otherwise selected from current production candidates):
AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL
AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL
AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME
AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS
EOF
}
[[ ${1:-} == --execute && $# -eq 2 ]] || {
usage >&2
exit 64
}
release_manifest=$2
[[ -f $release_manifest && ! -L $release_manifest ]] || {
echo 'release manifest must be a regular file' >&2
exit 1
}
load_cluster_env
require_commands curl git jq node openssl sed
: "${AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN:?}"
: "${AI_GATEWAY_ACCEPTANCE_API_KEY:?}"
: "${AI_GATEWAY_ACCEPTANCE_API_KEY_ID:?}"
: "${AI_GATEWAY_ACCEPTANCE_USER_ID:?}"
: "${AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL:=}"
: "${AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL:=}"
: "${AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS:=64}"
: "${AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS:=32}"
: "${AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME:=}"
: "${AI_GATEWAY_ACCEPTANCE_GATEWAYS:?}"
: "${AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS:?}"
[[ $AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS =~ ^[1-9][0-9]*$ &&
$AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS -le 256 ]] || {
echo 'AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS must be between 1 and 256' >&2
exit 1
}
[[ $AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS =~ ^[1-9][0-9]*$ &&
$AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS -le 128 ]] || {
echo 'AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS must be between 1 and 128' >&2
exit 1
}
node "$cluster_root/scripts/release-manifest.mjs" validate "$release_manifest" >/dev/null
release_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" sourceSha)
base_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" baseReleaseSha)
api_image=$(node "$cluster_root/scripts/release-manifest.mjs" get "$release_manifest" images.api)
[[ $release_sha =~ ^[0-9a-f]{40}$ && $api_image =~ @sha256:[0-9a-f]{64}$ ]] || {
echo 'release manifest is not full-SHA and digest pinned' >&2
exit 1
}
api_digest=${api_image##*@}
worker_digest=$api_digest
[[ $(git -C "$cluster_root" rev-parse HEAD) == "$release_sha" ]] || {
echo 'working copy HEAD must equal the release SHA' >&2
exit 1
}
[[ -z $(git -C "$cluster_root" status --short) ]] || {
echo 'production acceptance requires a clean release working copy' >&2
exit 1
}
namespace=${AI_GATEWAY_K3S_NAMESPACE:-easyai}
remote_release_helper=${AI_GATEWAY_REMOTE_RELEASE_HELPER:-/usr/local/sbin/easyai-ai-gateway-cluster-release}
[[ $namespace =~ ^[a-z0-9-]+$ && $remote_release_helper =~ ^/[A-Za-z0-9._/-]+$ ]] || {
echo 'invalid namespace or remote release helper' >&2
exit 1
}
[[ $namespace == easyai ]] || {
echo 'the production acceptance manifests currently require namespace easyai' >&2
exit 1
}
temporary_root=$(mktemp -d)
chmod 0700 "$temporary_root"
current_manifest=$temporary_root/current.json
run_token=$(openssl rand -hex 32)
run_id=
report_root=
stable_profile=P24
active_profile=P24
failure_reason=
failure_recorded=false
acceptance_admin_base=
acceptance_group_id=
acceptance_participants_json='[]'
AI_GATEWAY_ACCEPTANCE_API_KEYS=$AI_GATEWAY_ACCEPTANCE_API_KEY
cleanup() {
local status=$?
trap - EXIT
if (( status != 0 )) && [[ -n $run_id && $failure_recorded != true ]]; then
set +e
apply_capacity_profile "$stable_profile" >/dev/null 2>&1
mark_run_failed 'acceptance workflow exited unexpectedly'
set -e
fi
rm -rf -- "$temporary_root"
exit "$status"
}
trap cleanup EXIT
remote_kubectl() {
local command_text
printf -v command_text '%q ' "$@"
cluster_ssh "$CLUSTER_NINGBO_HOST" "k3s kubectl $command_text"
}
database_query() {
local sql=$1
local primary
primary=$(remote_kubectl get cluster easyai-postgres -n "$namespace" -o 'jsonpath={.status.currentPrimary}')
[[ $primary =~ ^easyai-postgres-[0-9]+$ ]]
remote_kubectl exec -n "$namespace" "$primary" -c postgres -- \
psql -X -v ON_ERROR_STOP=1 -U postgres -d easyai_ai_gateway -At -c "$sql"
}
admin_request() {
local method=$1
local path=$2
local body=${3:-}
local output=$4
local status response payload token_encoded body_encoded has_body=false
local service_ip service_port remote_script remote_command
if [[ -z $acceptance_admin_base ]]; then
service_ip=$(remote_kubectl get service api-ningbo-edge -n "$namespace" \
-o 'jsonpath={.spec.clusterIP}')
service_port=$(remote_kubectl get service api-ningbo-edge -n "$namespace" \
-o 'jsonpath={.spec.ports[0].port}')
[[ $service_ip =~ ^[0-9.]+$ && $service_port =~ ^[0-9]+$ ]] || {
echo 'acceptance admin Service has an invalid cluster address' >&2
return 1
}
acceptance_admin_base=http://$service_ip:$service_port
fi
token_encoded=$(printf '%s' "$AI_GATEWAY_ACCEPTANCE_ADMIN_TOKEN" |
openssl base64 -A)
body_encoded=$(printf '%s' "$body" | openssl base64 -A)
[[ -n $body ]] && has_body=true
# shellcheck disable=SC2016 # This script is quoted for execution on the remote node.
remote_script='
set -euo pipefail
admin_base=$1
method=$2
path=$3
token=$(printf "%s" "$4" | base64 -d)
body=$(printf "%s" "$5" | base64 -d)
has_body=$6
if [[ $has_body == true ]]; then
curl -sS --max-time 30 -w "\n%{http_code}" -X "$method" \
-H "Authorization: Bearer $token" \
-H "Content-Type: application/json" \
--data-binary "$body" "$admin_base$path"
else
curl -sS --max-time 30 -w "\n%{http_code}" -X "$method" \
-H "Authorization: Bearer $token" "$admin_base$path"
fi
'
printf -v remote_command 'bash -c %q -- %q %q %q %q %q %q' \
"$remote_script" "$acceptance_admin_base" "$method" "$path" \
"$token_encoded" "$body_encoded" "$has_body"
if ! response=$(cluster_ssh "$CLUSTER_NINGBO_HOST" "$remote_command"); then
echo "acceptance control API transport failed: method=$method path=$path" >&2
return 1
fi
status=${response##*$'\n'}
payload=${response%$'\n'*}
printf '%s' "$payload" >"$output"
[[ $status =~ ^2[0-9][0-9]$ ]] || {
echo "acceptance control API failed: method=$method path=$path status=$status" >&2
return 1
}
}
verify_release_cas() {
cluster_ssh "$CLUSTER_NINGBO_HOST" "$remote_release_helper status" >"$current_manifest"
node "$cluster_root/scripts/release-manifest.mjs" validate "$current_manifest" >/dev/null
local current_sha current_api
current_sha=$(node "$cluster_root/scripts/release-manifest.mjs" get "$current_manifest" sourceSha)
current_api=$(node "$cluster_root/scripts/release-manifest.mjs" get "$current_manifest" images.api)
[[ $current_sha == "$release_sha" && $current_api == "$api_image" ]] || {
echo "production release CAS mismatch: expected=$release_sha current=$current_sha" >&2
return 1
}
for deployment in easyai-api-ningbo easyai-api-hongkong easyai-worker-ningbo easyai-worker-hongkong; do
[[ $(remote_kubectl get deployment "$deployment" -n "$namespace" \
-o 'jsonpath={.spec.template.spec.containers[0].image}') == "$api_image" ]]
done
[[ $(remote_kubectl get deployment easyai-worker-ningbo -n "$namespace" \
-o 'jsonpath={.spec.replicas}') == 1 ]]
[[ $(remote_kubectl get deployment easyai-worker-hongkong -n "$namespace" \
-o 'jsonpath={.spec.replicas}') == 1 ]]
}
wait_for_existing_tasks_to_drain() {
local deadline=$((SECONDS + 900))
local active
while (( SECONDS < deadline )); do
active=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id IS NULL AND status IN ('queued','running');")
if [[ $active == 0 ]]; then
return 0
fi
echo "waiting_for_existing_tasks=$active"
sleep 5
done
echo 'existing production tasks did not drain within 15 minutes' >&2
return 1
}
wait_for_terminal_acceptance_tasks_to_drain() {
local deadline=$((SECONDS + 300))
local state active side_effects connections max_connections
while (( SECONDS < deadline )); do
state=$(database_query "
SELECT
(SELECT count(*)
FROM gateway_tasks task
JOIN gateway_acceptance_runs run ON run.id=task.acceptance_run_id
WHERE run.status IN ('failed', 'aborted')
AND task.status IN ('queued', 'running'))||','||
(
(SELECT count(*)
FROM settlement_outbox settlement
JOIN gateway_tasks task ON task.id=settlement.task_id
JOIN gateway_acceptance_runs run ON run.id=task.acceptance_run_id
WHERE run.status IN ('failed', 'aborted')
AND settlement.status IN ('pending', 'processing', 'retryable_failed'))
+
(SELECT count(*)
FROM gateway_task_callback_outbox callback
JOIN gateway_tasks task ON task.id=callback.task_id
JOIN gateway_acceptance_runs run ON run.id=task.acceptance_run_id
WHERE run.status IN ('failed', 'aborted')
AND callback.status <> 'delivered')
)||','||
(SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend')||','||
(SELECT setting FROM pg_settings WHERE name='max_connections');")
IFS=',' read -r active side_effects connections max_connections <<<"$state"
if [[ $active == 0 && $side_effects == 0 ]] &&
(( connections * 2 < max_connections )); then
return 0
fi
echo "waiting_for_terminal_acceptance_tasks=$active side_effects=$side_effects database_connections=$connections"
sleep 2
done
echo "terminal acceptance state did not drain within 5 minutes: active=$active side_effects=$side_effects database_connections=$connections" >&2
return 1
}
ensure_acceptance_user_group() {
local groups_response=$temporary_root/user-groups.json
local group_response=$temporary_root/acceptance-user-group.json
local users_response=$temporary_root/users.json
local user_response=$temporary_root/acceptance-user.json
local group_id group_body user_body isolation_state
admin_request GET /api/admin/user-groups '' "$groups_response"
group_id=$(jq -r '
[.items[] | select(.groupKey == "production-acceptance") | .id]
| if length == 1 then .[0] else empty end
' "$groups_response")
group_body=$(jq -cn '{
groupKey:"production-acceptance",
name:"Production Acceptance",
description:"Dedicated production-isomorphic acceptance traffic",
source:"gateway",
priority:1,
rechargeDiscountPolicy:{},
billingDiscountPolicy:{},
rateLimitPolicy:{rules:[]},
quotaPolicy:{},
metadata:{purpose:"production_acceptance",isolated:true},
status:"active"
}')
if [[ -z $group_id ]]; then
[[ $(jq '[.items[] | select(.groupKey == "production-acceptance")] | length' \
"$groups_response") == 0 ]] || {
echo 'acceptance user group lookup was ambiguous' >&2
return 1
}
admin_request POST /api/admin/user-groups "$group_body" "$group_response"
group_id=$(jq -r '.id // empty' "$group_response")
else
[[ $group_id =~ ^[0-9a-f-]{36}$ ]] || {
echo 'acceptance user group API returned an invalid group ID' >&2
return 1
}
isolation_state=$(database_query "
SELECT
(SELECT count(*)
FROM gateway_users
WHERE default_user_group_id='$group_id'::uuid
AND id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND COALESCE(metadata->>'purpose','') <> 'production_acceptance_shard')
||':'||
(SELECT count(*)
FROM gateway_api_keys key
JOIN gateway_users user_record ON user_record.id=key.gateway_user_id
WHERE key.user_group_id='$group_id'::uuid
AND key.gateway_user_id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND COALESCE(user_record.metadata->>'purpose','') <> 'production_acceptance_shard');")
[[ $isolation_state == 0:0 ]] || {
echo 'acceptance user group is referenced by non-acceptance identities' >&2
return 1
}
admin_request PATCH "/api/admin/user-groups/$group_id" \
"$group_body" "$group_response"
fi
[[ $group_id =~ ^[0-9a-f-]{36}$ ]] || {
echo 'acceptance user group API returned an invalid group ID' >&2
return 1
}
acceptance_group_id=$group_id
admin_request GET /api/admin/users '' "$users_response"
user_body=$(jq -c \
--arg userID "$AI_GATEWAY_ACCEPTANCE_USER_ID" \
--arg groupID "$group_id" '
[.items[] | select(.id == $userID)]
| if length != 1 then empty else .[0] end
| {
userKey:(.userKey // ""),
source:(.source // "gateway"),
externalUserId:(.externalUserId // ""),
username:(.username // ""),
displayName:(.displayName // ""),
email:(.email // ""),
phone:(.phone // ""),
avatarUrl:(.avatarUrl // ""),
gatewayTenantId:(.gatewayTenantId // ""),
tenantId:(.tenantId // ""),
tenantKey:(.tenantKey // ""),
defaultUserGroupId:$groupID,
roles:(.roles // []),
authProfile:(.authProfile // {}),
metadata:(.metadata // {}),
status:(.status // "active")
}
' "$users_response")
[[ -n $user_body && $(jq -r '.username // empty' <<<"$user_body") != '' ]] || {
echo 'dedicated acceptance user was not found through the control API' >&2
return 1
}
admin_request PATCH "/api/admin/users/$AI_GATEWAY_ACCEPTANCE_USER_ID" \
"$user_body" "$user_response"
isolation_state=$(database_query "
SELECT
(SELECT count(*)
FROM gateway_users
WHERE id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND default_user_group_id='$group_id'::uuid)
||':'||
(SELECT count(*)
FROM gateway_api_keys
WHERE id='$AI_GATEWAY_ACCEPTANCE_API_KEY_ID'::uuid
AND gateway_user_id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND COALESCE(user_group_id,'$group_id'::uuid)='$group_id'::uuid
AND status='active'
AND deleted_at IS NULL)
||':'||
(SELECT count(*)
FROM gateway_api_keys
WHERE gateway_user_id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND status='active'
AND deleted_at IS NULL)
||':'||
(SELECT count(*)
FROM gateway_user_groups
WHERE id='$group_id'::uuid
AND status='active'
AND NOT EXISTS (
SELECT 1
FROM jsonb_array_elements(
COALESCE(rate_limit_policy->'rules','[]'::jsonb)
) rule
WHERE rule->>'metric' IN ('concurrent','queue_size')
))
||':'||
(SELECT count(*)
FROM gateway_users
WHERE default_user_group_id='$group_id'::uuid
AND id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND COALESCE(metadata->>'purpose','') <> 'production_acceptance_shard')
||':'||
(SELECT count(*)
FROM gateway_api_keys key
JOIN gateway_users user_record ON user_record.id=key.gateway_user_id
WHERE key.user_group_id='$group_id'::uuid
AND key.gateway_user_id <> '$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND COALESCE(user_record.metadata->>'purpose','') <> 'production_acceptance_shard');")
[[ $isolation_state == 1:1:1:1:0:0 ]] || {
echo 'dedicated acceptance identity isolation verification failed' >&2
return 1
}
echo 'acceptance_user_group=PASS rate_limit=production_candidates active_keys=1 isolated=true'
}
ensure_acceptance_identity_shards() {
local identity_material key_count participant_count
[[ $acceptance_group_id =~ ^[0-9a-f-]{36}$ ]] || {
echo 'acceptance user group must be prepared before identity shards' >&2
return 1
}
identity_material=$(database_query "
WITH primary_user AS (
SELECT roles
FROM gateway_users
WHERE id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND status='active'
AND deleted_at IS NULL
), desired AS (
SELECT ordinal,
'production-acceptance-shard-' || lpad(ordinal::text, 3, '0') AS user_key
FROM generate_series(1, $((AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS - 1))) ordinal
), shard_users AS (
INSERT INTO gateway_users (
user_key, source, username, display_name, default_user_group_id,
roles, auth_profile, metadata, status
)
SELECT desired.user_key,
'gateway',
desired.user_key,
'Production Acceptance Shard ' || lpad(desired.ordinal::text, 3, '0'),
'$acceptance_group_id'::uuid,
primary_user.roles,
'{}'::jsonb,
jsonb_build_object(
'purpose', 'production_acceptance_shard',
'ordinal', desired.ordinal,
'isolated', true
),
'active'
FROM desired
CROSS JOIN primary_user
ON CONFLICT (user_key) DO UPDATE
SET default_user_group_id=EXCLUDED.default_user_group_id,
roles=EXCLUDED.roles,
metadata=EXCLUDED.metadata,
status='active',
deleted_at=NULL,
updated_at=now()
RETURNING id, user_key
), primary_key AS (
SELECT id, gateway_user_id, key_secret, scopes, rate_limit_policy, quota_policy
FROM gateway_api_keys
WHERE id='$AI_GATEWAY_ACCEPTANCE_API_KEY_ID'::uuid
AND gateway_user_id='$AI_GATEWAY_ACCEPTANCE_USER_ID'::uuid
AND status='active'
AND deleted_at IS NULL
AND COALESCE(key_secret, '') <> ''
), existing_keys AS (
SELECT selected.id, selected.gateway_user_id, selected.key_secret
FROM shard_users shard
JOIN LATERAL (
SELECT key.id, key.gateway_user_id, key.key_secret
FROM gateway_api_keys key
WHERE key.gateway_user_id=shard.id
AND key.name='Production Acceptance Shard'
AND key.status='active'
AND key.deleted_at IS NULL
AND COALESCE(key.key_secret, '') <> ''
ORDER BY key.created_at
LIMIT 1
) selected ON true
), new_key_material AS (
SELECT shard.id AS gateway_user_id,
'sk-gw-' || encode(gen_random_bytes(32), 'hex') AS secret
FROM shard_users shard
WHERE NOT EXISTS (
SELECT 1 FROM existing_keys existing
WHERE existing.gateway_user_id=shard.id
)
), inserted_keys AS (
INSERT INTO gateway_api_keys (
gateway_user_id, user_id, key_prefix, key_secret, key_hash, name, scopes,
user_group_id, rate_limit_policy, quota_policy, status
)
SELECT material.gateway_user_id,
material.gateway_user_id::text,
left(material.secret, 18),
material.secret,
crypt(material.secret, gen_salt('bf', 10)),
'Production Acceptance Shard',
template.scopes,
'$acceptance_group_id'::uuid,
template.rate_limit_policy,
template.quota_policy,
'active'
FROM new_key_material material
CROSS JOIN primary_key template
RETURNING id, gateway_user_id, key_secret
), shard_keys AS (
SELECT * FROM existing_keys
UNION ALL
SELECT * FROM inserted_keys
), wallets AS (
INSERT INTO gateway_wallet_accounts (
gateway_user_id, user_id, currency, balance, total_recharged, metadata, status
)
SELECT shard.id,
shard.id::text,
'resource',
1000000000::numeric,
1000000000::numeric,
'{\"purpose\":\"production_acceptance_shard\",\"isolated\":true}'::jsonb,
'active'
FROM shard_users shard
ON CONFLICT (gateway_user_id, currency) DO UPDATE
SET balance=GREATEST(gateway_wallet_accounts.balance, EXCLUDED.balance),
total_recharged=GREATEST(gateway_wallet_accounts.total_recharged, EXCLUDED.total_recharged),
metadata=EXCLUDED.metadata,
status='active',
updated_at=now()
RETURNING gateway_user_id
), participants AS (
SELECT 0 AS ordinal, key.id, key.gateway_user_id, key.key_secret
FROM primary_key key
UNION ALL
SELECT row_number() OVER (ORDER BY shard.user_key)::int,
key.id,
key.gateway_user_id,
key.key_secret
FROM shard_keys key
JOIN shard_users shard ON shard.id=key.gateway_user_id
)
SELECT jsonb_agg(
jsonb_build_object(
'apiKeyId', participant.id,
'userId', participant.gateway_user_id
)
ORDER BY participant.ordinal
)::text
|| E'\\t' ||
string_agg(participant.key_secret, ',' ORDER BY participant.ordinal)
FROM participants participant
CROSS JOIN (SELECT count(*) FROM wallets) wallet_write_barrier;")
[[ $identity_material == *$'\t'* ]] || {
echo 'acceptance identity shard provisioning returned an invalid payload' >&2
return 1
}
acceptance_participants_json=${identity_material%%$'\t'*}
AI_GATEWAY_ACCEPTANCE_API_KEYS=${identity_material#*$'\t'}
participant_count=$(jq 'length' <<<"$acceptance_participants_json")
key_count=$(awk -F',' '{print NF}' <<<"$AI_GATEWAY_ACCEPTANCE_API_KEYS")
[[ $participant_count == "$AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS" &&
$key_count == "$AI_GATEWAY_ACCEPTANCE_IDENTITY_SHARDS" ]] || {
echo 'acceptance identity shard count verification failed' >&2
return 1
}
export AI_GATEWAY_ACCEPTANCE_API_KEYS
echo "acceptance_identity_shards=PASS count=$participant_count isolated_wallets=true"
}
select_acceptance_models() {
if [[ -z $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL ]]; then
AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL=$(database_query "
SELECT b.invocation_name
FROM platform_models model
JOIN integration_platforms platform ON platform.id=model.platform_id
JOIN base_model_catalog b ON b.id=model.base_model_id
WHERE platform.status='enabled' AND platform.deleted_at IS NULL
AND model.enabled=true
AND lower(platform.provider) IN ('gemini','google-gemini','gemini-openai')
AND model.model_type @> '[\"image_edit\"]'::jsonb
ORDER BY
CASE
WHEN lower(b.invocation_name) = 'gemini-3.1-flash-image' THEN 0
WHEN lower(b.invocation_name) LIKE '%gemini%image%'
AND lower(b.invocation_name) NOT LIKE '%preview%' THEN 1
WHEN lower(b.invocation_name) LIKE '%gemini%image%' THEN 2
ELSE 3
END,
platform.priority, model.created_at
LIMIT 1;")
fi
if [[ -z $AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL ]]; then
AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL=$(database_query "
SELECT b.invocation_name
FROM platform_models model
JOIN integration_platforms platform ON platform.id=model.platform_id
JOIN base_model_catalog b ON b.id=model.base_model_id
WHERE platform.status='enabled' AND platform.deleted_at IS NULL
AND model.enabled=true
AND model.model_type @> '[\"omni_video\"]'::jsonb
AND COALESCE(
NULLIF(model.capability_override #>> '{omni_video,max_images}','')::int,
NULLIF(model.capabilities #>> '{omni_video,max_images}','')::int,
NULLIF(b.capabilities #>> '{omni_video,max_images}','')::int,
0
) >= 9
ORDER BY
CASE
WHEN lower(b.invocation_name) LIKE '%seedance%2%fast%' THEN 0
WHEN lower(b.invocation_name) LIKE '%seedance%2%' THEN 1
ELSE 2
END,
platform.priority, model.created_at
LIMIT 1;")
fi
[[ -n $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL && -n $AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL ]] || {
echo 'no enabled Gemini image-edit or omni-video max_images>=9 model was found' >&2
return 1
}
[[ $AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL =~ ^[A-Za-z0-9._:/+-]+$ &&
$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL =~ ^[A-Za-z0-9._:/+-]+$ ]] || {
echo 'acceptance model invocation names contain unsupported characters' >&2
return 1
}
export AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL
}
ensure_acceptance_model_access() {
local resources response body missing expected
response=$temporary_root/model-access.json
resources=$(database_query "
WITH selected_candidates AS (
SELECT DISTINCT platform.id AS platform_id, model.id AS platform_model_id, model.base_model_id
FROM platform_models model
JOIN integration_platforms platform ON platform.id=model.platform_id
JOIN base_model_catalog base_model ON base_model.id=model.base_model_id
WHERE platform.status='enabled'
AND platform.deleted_at IS NULL
AND model.enabled=true
AND base_model.invocation_name IN (
'$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL',
'$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL'
)
), resources AS (
SELECT 'platform'::text AS resource_type, platform_id AS resource_id
FROM selected_candidates
UNION
SELECT 'platform_model', platform_model_id
FROM selected_candidates
UNION
SELECT 'base_model', base_model_id
FROM selected_candidates
WHERE base_model_id IS NOT NULL
)
SELECT COALESCE(
jsonb_agg(
jsonb_build_object(
'resourceType', resource_type,
'resourceId', resource_id,
'priority', 100,
'minPermissionLevel', 0,
'status', 'active'
)
ORDER BY resource_type, resource_id
),
'[]'::jsonb
)::text
FROM resources;")
expected=$(jq 'length' <<<"$resources")
(( expected > 0 )) || {
echo 'selected acceptance models did not resolve to production candidate resources' >&2
return 1
}
body=$(jq -cn \
--arg subjectID "$acceptance_group_id" \
--argjson resources "$resources" \
'{
subjectType:"user_group",
subjectId:$subjectID,
effect:"allow",
upsertResources:$resources,
deleteResources:[]
}')
admin_request POST /api/admin/access-rules/batch "$body" "$response"
missing=$(database_query "
WITH selected_candidates AS (
SELECT DISTINCT platform.id AS platform_id, model.id AS platform_model_id, model.base_model_id
FROM platform_models model
JOIN integration_platforms platform ON platform.id=model.platform_id
JOIN base_model_catalog base_model ON base_model.id=model.base_model_id
WHERE platform.status='enabled'
AND platform.deleted_at IS NULL
AND model.enabled=true
AND base_model.invocation_name IN (
'$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL',
'$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL'
)
), resources AS (
SELECT 'platform'::text AS resource_type, platform_id AS resource_id
FROM selected_candidates
UNION
SELECT 'platform_model', platform_model_id
FROM selected_candidates
UNION
SELECT 'base_model', base_model_id
FROM selected_candidates
WHERE base_model_id IS NOT NULL
)
SELECT count(*)
FROM resources resource
WHERE NOT EXISTS (
SELECT 1
FROM gateway_access_rules rule
WHERE rule.subject_type='user_group'
AND rule.subject_id='$acceptance_group_id'::uuid
AND rule.effect='allow'
AND rule.status='active'
AND rule.resource_type=resource.resource_type
AND rule.resource_id=resource.resource_id
);")
[[ $missing == 0 ]] || {
echo "acceptance user group is missing $missing selected candidate access rules" >&2
return 1
}
echo "acceptance_model_access=PASS resources=$expected"
}
deploy_protocol_emulator() {
sed "s|image: easyai-api|image: $api_image|" \
"$cluster_root/deploy/kubernetes/acceptance/protocol-emulator.yaml" |
cluster_ssh "$CLUSTER_NINGBO_HOST" 'k3s kubectl apply -f -' >/dev/null
remote_kubectl rollout status deployment/easyai-acceptance-emulator \
-n "$namespace" --timeout=300s
}
create_and_activate_run() {
local create_response=$temporary_root/create-run.json
local activate_response=$temporary_root/activate-run.json
local traffic_response=$temporary_root/pre-activate-traffic.json
local previous_run_response=$temporary_root/previous-run.json
local abort_response=$temporary_root/abort-previous-run.json
local body
body=$(jq -cn \
--arg releaseSha "$release_sha" \
--arg apiDigest "$api_digest" \
--arg workerDigest "$worker_digest" \
--arg apiKeyID "$AI_GATEWAY_ACCEPTANCE_API_KEY_ID" \
--arg userID "$AI_GATEWAY_ACCEPTANCE_USER_ID" \
--arg token "$run_token" \
--arg emulatorURL 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090' \
--arg callbackURL 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090/callbacks' \
--argjson participants "$acceptance_participants_json" \
'{
releaseSha: $releaseSha,
apiImageDigest: $apiDigest,
workerImageDigest: $workerDigest,
apiKeyId: $apiKeyID,
userId: $userID,
token: $token,
emulatorBaseUrl: $emulatorURL,
callbackUrl: $callbackURL,
capacityProfile: "P24",
config: {
workloads: ["gemini_image_edit", "multi_reference_video"],
participants: $participants
}
}')
admin_request POST /api/admin/system/acceptance/runs "$body" "$create_response"
run_id=$(jq -r '.id // empty' "$create_response")
[[ $run_id =~ ^[0-9a-f-]{36}$ ]] || {
echo 'acceptance control API returned an invalid Run ID' >&2
return 1
}
report_root=${AI_GATEWAY_ACCEPTANCE_REPORT_DIR:-"$cluster_root/dist/acceptance/$run_id"}
mkdir -p "$report_root"
chmod 0700 "$report_root"
admin_request GET /api/admin/system/acceptance/traffic-mode '' "$traffic_response"
local traffic_mode previous_run_id previous_status
local traffic_release traffic_api_digest traffic_worker_digest
local previous_release previous_api_digest previous_worker_digest
traffic_mode=$(jq -r '.mode // empty' "$traffic_response")
if [[ $traffic_mode == validation ]]; then
previous_run_id=$(jq -r '.runId // empty' "$traffic_response")
[[ $previous_run_id =~ ^[0-9a-f-]{36}$ && $previous_run_id != "$run_id" ]] || {
echo 'validation mode does not reference a different previous Run' >&2
return 1
}
admin_request GET "/api/admin/system/acceptance/runs/$previous_run_id" '' \
"$previous_run_response"
previous_status=$(jq -r '.status // empty' "$previous_run_response")
[[ $previous_status == failed ]] || {
echo "refusing to replace previous acceptance Run in status $previous_status" >&2
return 1
}
traffic_release=$(jq -r '.releaseSha // empty' "$traffic_response")
traffic_api_digest=$(jq -r '.apiImageDigest // empty' "$traffic_response")
traffic_worker_digest=$(jq -r '.workerImageDigest // empty' "$traffic_response")
previous_release=$(jq -r '.releaseSha // empty' "$previous_run_response")
previous_api_digest=$(jq -r '.apiImageDigest // empty' "$previous_run_response")
previous_worker_digest=$(jq -r '.workerImageDigest // empty' "$previous_run_response")
[[ $traffic_release == "$previous_release" &&
$traffic_api_digest == "$previous_api_digest" &&
$traffic_worker_digest == "$previous_worker_digest" ]] || {
echo 'previous failed Run does not match the active validation CAS fields' >&2
return 1
}
if [[ $traffic_release == "$release_sha" ]]; then
[[ $traffic_api_digest == "$api_digest" &&
$traffic_worker_digest == "$worker_digest" ]] || {
echo 'current-release validation mode has unexpected image digests' >&2
return 1
}
elif [[ -z $base_sha || $traffic_release != "$base_sha" ]]; then
echo 'refusing to replace validation mode outside the direct release base' >&2
return 1
fi
body=$(jq -cn \
--argjson revision "$(jq -r '.revision' "$traffic_response")" \
--arg releaseSha "$traffic_release" \
--arg apiDigest "$traffic_api_digest" \
--arg workerDigest "$traffic_worker_digest" \
'{
revision:$revision,
releaseSha:$releaseSha,
apiImageDigest:$apiDigest,
workerImageDigest:$workerDigest
}')
admin_request POST \
"/api/admin/system/acceptance/runs/$previous_run_id/abort" \
"$body" "$abort_response"
[[ $(jq -r '.mode // empty' "$abort_response") == live ]] || {
echo 'previous failed acceptance Run did not return traffic to live' >&2
return 1
}
wait_for_terminal_acceptance_tasks_to_drain
elif [[ $traffic_mode != live ]]; then
echo "unsupported production traffic mode before acceptance: $traffic_mode" >&2
return 1
fi
admin_request POST "/api/admin/system/acceptance/runs/$run_id/activate" '' "$activate_response"
[[ $(jq -r '.mode' "$activate_response") == validation &&
$(jq -r '.runId' "$activate_response") == "$run_id" ]]
}
mark_run_failed() {
local reason=$1
local response=$temporary_root/failed-run.json
local body
[[ -n $run_id ]] || return 0
body=$(jq -cn --arg reason "$reason" \
'{passed:false,failureReason:$reason,report:{queueFinal:null}}')
if admin_request POST "/api/admin/system/acceptance/runs/$run_id/finish" "$body" "$response"; then
failure_recorded=true
fi
}
apply_capacity_profile() {
local profile=$1
local slots pool media global
local api_pool=$AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS
local min_idle=4
local max_idle_seconds=30
case $profile in
P24) slots=24; pool=32; media=24; global=48 ;;
P28) slots=28; pool=36; media=28; global=56 ;;
P32) slots=32; pool=40; media=32; global=64 ;;
*) return 1 ;;
esac
active_profile=$profile
for site in ningbo hongkong; do
remote_kubectl set env "deployment/easyai-worker-$site" -n "$namespace" \
"AI_GATEWAY_ASYNC_WORKER_INSTANCE_HARD_LIMIT=$slots" \
"AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT=$global" \
"AI_GATEWAY_DATABASE_MAX_CONNS=$pool" \
"AI_GATEWAY_DATABASE_MIN_IDLE_CONNS=$min_idle" \
"AI_GATEWAY_DATABASE_MAX_CONN_IDLE_SECONDS=$max_idle_seconds" \
"AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY=$media" \
"AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY=$media" >/dev/null
remote_kubectl rollout status "deployment/easyai-worker-$site" \
-n "$namespace" --timeout=300s
remote_kubectl set env "deployment/easyai-api-$site" -n "$namespace" \
"AI_GATEWAY_ASYNC_WORKER_HARD_LIMIT=$global" \
"AI_GATEWAY_DATABASE_MAX_CONNS=$api_pool" \
"AI_GATEWAY_DATABASE_MIN_IDLE_CONNS=$min_idle" \
"AI_GATEWAY_DATABASE_MAX_CONN_IDLE_SECONDS=$max_idle_seconds" \
"AI_GATEWAY_MEDIA_MATERIALIZATION_CONCURRENCY=$media" \
"AI_GATEWAY_MEDIA_REQUEST_CONCURRENCY=$media" >/dev/null
remote_kubectl rollout status "deployment/easyai-api-$site" \
-n "$namespace" --timeout=300s
done
}
run_load_profile() {
local profile=$1
local report_path=$2
local load_pod remote_script
[[ $profile =~ ^[a-z-]+$ ]] || return 1
load_pod=$(remote_kubectl get pods -n "$namespace" \
-l 'app.kubernetes.io/name=easyai-acceptance-emulator' -o json |
jq -r '[.items[]
| select(any(.status.conditions[]?; .type=="Ready" and .status=="True"))
| .metadata.name][0] // empty')
[[ $load_pod =~ ^easyai-acceptance-emulator-[a-z0-9-]+$ ]] || {
echo 'acceptance load runner Pod is not Ready' >&2
return 1
}
# Feed private values through stdin so the API Key and Run Token never enter
# the Pod spec, remote command arguments, process list, or report.
# shellcheck disable=SC2016 # This script runs inside the release image.
remote_script='
set -eu
read -r gateways_b64
read -r tls_name_b64
read -r emulator_b64
read -r api_keys_b64
read -r run_id_b64
read -r run_token_b64
read -r gemini_model_b64
read -r video_model_b64
read -r image_urls_b64
decode() {
printf "%s" "$1" | base64 -d
}
export AI_GATEWAY_ACCEPTANCE_GATEWAYS=$(decode "$gateways_b64")
export AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME=$(decode "$tls_name_b64")
export AI_GATEWAY_ACCEPTANCE_EMULATOR_URL=$(decode "$emulator_b64")
export AI_GATEWAY_ACCEPTANCE_API_KEYS=$(decode "$api_keys_b64")
export AI_GATEWAY_ACCEPTANCE_RUN_ID=$(decode "$run_id_b64")
export AI_GATEWAY_ACCEPTANCE_RUN_TOKEN=$(decode "$run_token_b64")
export AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL=$(decode "$gemini_model_b64")
export AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL=$(decode "$video_model_b64")
export AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS=$(decode "$image_urls_b64")
set +e
report=$(/app/easyai-ai-gateway-acceptance-load -profile "$1")
status=$?
set -e
printf "%s\n" "$report"
exit "$status"
'
{
printf '%s' "$AI_GATEWAY_ACCEPTANCE_GATEWAYS" | openssl base64 -A
printf '\n'
printf '%s' "$AI_GATEWAY_ACCEPTANCE_GATEWAY_TLS_SERVER_NAME" | openssl base64 -A
printf '\n'
printf '%s' 'http://easyai-acceptance-emulator.easyai.svc.cluster.local:8090' |
openssl base64 -A
printf '\n'
printf '%s' "$AI_GATEWAY_ACCEPTANCE_API_KEYS" | openssl base64 -A
printf '\n'
printf '%s' "$run_id" | openssl base64 -A
printf '\n'
printf '%s' "$run_token" | openssl base64 -A
printf '\n'
printf '%s' "$AI_GATEWAY_ACCEPTANCE_GEMINI_MODEL" | openssl base64 -A
printf '\n'
printf '%s' "$AI_GATEWAY_ACCEPTANCE_VIDEO_MODEL" | openssl base64 -A
printf '\n'
printf '%s' "$AI_GATEWAY_ACCEPTANCE_REAL_IMAGE_URLS" | openssl base64 -A
printf '\n'
} |
remote_kubectl exec -i -n "$namespace" "$load_pod" -- \
sh -c "$remote_script" -- "$profile" >"$report_path"
}
kill_one_worker_during_recovery() {
local deadline=$((SECONDS + 120))
local running pod
while (( SECONDS < deadline )); do
running=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND status='running' AND request::text LIKE '%acceptance-long-recovery%';")
if (( running > 0 )); then
break
fi
sleep 1
done
(( running > 0 )) || {
echo 'recovery workload never reached a running Worker' >&2
return 1
}
pod=$(remote_kubectl get pods -n "$namespace" \
-l 'app.kubernetes.io/name=easyai-worker,easyai.io/site=hongkong' \
-o 'jsonpath={.items[0].metadata.name}')
[[ $pod =~ ^easyai-worker-hongkong-[a-z0-9-]+$ ]]
remote_kubectl delete pod "$pod" -n "$namespace" --wait=false >/dev/null
remote_kubectl rollout status deployment/easyai-worker-hongkong \
-n "$namespace" --timeout=300s
}
run_recovery_profile() {
local report_path=$1
run_load_profile video-recovery "$report_path" &
local load_pid=$!
if ! kill_one_worker_during_recovery; then
kill "$load_pid" >/dev/null 2>&1 || true
wait "$load_pid" >/dev/null 2>&1 || true
return 1
fi
wait "$load_pid"
}
verify_wireguard() {
local source_host=$1
local target_ip=$2
local max_rtt=$3
local label=$4
local output loss average handshakes now latest age
output=$(cluster_ssh "$source_host" "ping -q -c 10 -W 2 $target_ip")
loss=$(awk -F',' '/packet loss/ {gsub(/[^0-9.]/, "", $3); print $3}' <<<"$output")
average=$(awk -F'=' '/min\\/avg\\/max/ {split($2, values, \"/\"); gsub(/ /, "", values[2]); print values[2]}' <<<"$output")
[[ -n $loss && -n $average ]]
awk -v loss="$loss" -v average="$average" -v max="$max_rtt" \
'BEGIN { exit !(loss < 1 && average < max) }'
now=$(date +%s)
latest=$(cluster_ssh "$source_host" "wg show all latest-handshakes | awk '{print \\$3}' | sort -nr | head -n 2")
handshakes=$(wc -l <<<"$latest" | tr -d ' ')
(( handshakes >= 2 ))
while IFS= read -r timestamp; do
[[ $timestamp =~ ^[0-9]+$ && $timestamp -gt 0 ]]
age=$((now - timestamp))
(( age < 180 ))
done <<<"$latest"
echo "wireguard=$label loss_percent=$loss average_rtt_ms=$average recent_handshakes=$handshakes"
}
verify_cluster_links() {
verify_wireguard "$CLUSTER_NINGBO_HOST" 10.77.0.2 80 'ningbo_to_hongkong'
verify_wireguard "$CLUSTER_HONGKONG_HOST" 10.77.0.1 80 'hongkong_to_ningbo'
verify_wireguard "$CLUSTER_NINGBO_HOST" 10.77.0.3 300 'ningbo_to_los_angeles'
verify_wireguard "$CLUSTER_LOS_ANGELES_HOST" 10.77.0.1 300 'los_angeles_to_ningbo'
verify_wireguard "$CLUSTER_HONGKONG_HOST" 10.77.0.3 300 'hongkong_to_los_angeles'
verify_wireguard "$CLUSTER_LOS_ANGELES_HOST" 10.77.0.2 300 'los_angeles_to_hongkong'
}
verify_control_plane_and_recent_logs() {
local host ready_output workload pod error_count
for host in "$CLUSTER_NINGBO_HOST" "$CLUSTER_HONGKONG_HOST" "$CLUSTER_LOS_ANGELES_HOST"; do
ready_output=$(cluster_ssh "$host" "k3s kubectl get --raw='/readyz?verbose'")
grep -Fq '[+]etcd ok' <<<"$ready_output"
done
for workload in api worker; do
while read -r pod; do
error_count=$(remote_kubectl logs "$pod" -n "$namespace" --since=10m |
awk 'BEGIN {IGNORECASE=1}
/postgres_unavailable/ ||
/postgres readiness.*(fail|unavailable|error)/ ||
/leadership elector.*(error|fail)/ ||
/concurrency lease.*(renew.*(error|fail)|lost)/ {count++}
END {print count+0}')
[[ $error_count == 0 ]]
done < <(remote_kubectl get pods -n "$namespace" \
-l "app.kubernetes.io/name=easyai-$workload" \
-o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}')
done
while read -r pod; do
error_count=$(remote_kubectl logs "$pod" -n "$namespace" -c postgres --since=10m |
awk 'BEGIN {IGNORECASE=1}
/apply request took too long/ ||
/dial tcp.*6443/ ||
/synchronous commit.*cancel/ ||
/canceling statement due to user request/ ||
/(liveness|readiness).*fail/ {count++}
END {print count+0}')
[[ $error_count == 0 ]]
done < <(remote_kubectl get pods -n "$namespace" \
-l 'cnpg.io/cluster=easyai-postgres' \
-o jsonpath='{range .items[*]}{.metadata.name}{"\n"}{end}')
}
current_max_pod_memory_mib() {
remote_kubectl top pods -n "$namespace" \
-l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers |
awk '{
value=$3
if (value ~ /Gi$/) {sub(/Gi$/, "", value); value*=1024}
else if (value ~ /Mi$/) {sub(/Mi$/, "", value)}
else if (value ~ /Ki$/) {sub(/Ki$/, "", value); value/=1024}
if (value>max) max=value
} END {printf "%.0f", max+0}'
}
wait_for_memory_recovery() {
local baseline_mib=$1
local deadline=$((SECONDS + 120))
local current_mib
while (( SECONDS < deadline )); do
current_mib=$(current_max_pod_memory_mib)
if (( current_mib <= baseline_mib + 256 )); then
return 0
fi
sleep 5
done
echo "Pod memory did not return near baseline: baseline_mib=$baseline_mib current_mib=$current_mib" >&2
return 1
}
verify_runtime_gates() {
local expected_slots expected_global expected_pool
local expected_api_pool=$AI_GATEWAY_ACCEPTANCE_API_DATABASE_MAX_CONNS
case $active_profile in
P24) expected_slots=24; expected_pool=32; expected_global=48 ;;
P28) expected_slots=28; expected_pool=36; expected_global=56 ;;
P32) expected_slots=32; expected_pool=40; expected_global=64 ;;
esac
[[ $(remote_kubectl get nodes -o json |
jq '[.items[] | select(any(.status.conditions[]; .type=="Ready" and .status=="True"))] | length') == 3 ]]
[[ $(remote_kubectl get nodes -o json |
jq '[.items[] | select(any(.status.conditions[]; .type=="MemoryPressure" and .status=="True"))] | length') == 0 ]]
[[ $(remote_kubectl get cluster easyai-postgres -n "$namespace" -o 'jsonpath={.status.readyInstances}') == 2 ]]
local sync_state
sync_state=$(database_query "SELECT COALESCE(string_agg(sync_state,','),'') FROM pg_stat_replication;")
[[ ",$sync_state," == *,sync,* || ",$sync_state," == *,quorum,* ]]
verify_control_plane_and_recent_logs
local unhealthy_containers pod_memory
unhealthy_containers=$(remote_kubectl get pods -n "$namespace" -o json |
jq '[.items[]
| select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or
.metadata.labels["app.kubernetes.io/name"] == "easyai-worker")
| .status.containerStatuses[]?
| select(.restartCount > 0 or .lastState.terminated.reason == "OOMKilled" or
.state.terminated.reason == "OOMKilled")] | length')
[[ $unhealthy_containers == 0 ]]
while read -r _node _cpu _cpu_percent _memory memory_percent; do
memory_percent=${memory_percent%\%}
[[ $memory_percent =~ ^[0-9]+$ ]]
(( memory_percent < 80 ))
done < <(remote_kubectl top nodes --no-headers)
while read -r _pod _cpu pod_memory; do
case $pod_memory in
*Gi) pod_memory=$(awk -v value="${pod_memory%Gi}" 'BEGIN { printf "%.0f", value * 1024 }') ;;
*Mi) pod_memory=${pod_memory%Mi} ;;
*Ki) pod_memory=$(awk -v value="${pod_memory%Ki}" 'BEGIN { printf "%.0f", value / 1024 }') ;;
*) return 1 ;;
esac
[[ $pod_memory =~ ^[0-9]+$ ]]
(( pod_memory < 1536 ))
done < <(remote_kubectl top pods -n "$namespace" \
-l 'app.kubernetes.io/part-of=easyai-ai-gateway' --no-headers)
local pod_name pod_anon_mib
while read -r pod_name; do
pod_anon_mib=$(remote_kubectl exec -n "$namespace" "$pod_name" -- \
cat /sys/fs/cgroup/memory.stat |
awk '$1=="anon" {printf "%.0f", $2/1048576}')
[[ $pod_anon_mib =~ ^[0-9]+$ ]]
(( pod_anon_mib < 1536 ))
done < <(remote_kubectl get pods -n "$namespace" -o json |
jq -r '.items[]
| select(.metadata.labels["app.kubernetes.io/name"] == "easyai-api" or
.metadata.labels["app.kubernetes.io/name"] == "easyai-worker")
| .metadata.name')
local completion_deadline=$((SECONDS + 180))
local incomplete side_effects
while (( SECONDS < completion_deadline )); do
incomplete=$(database_query "
SELECT count(*)
FROM gateway_tasks
WHERE acceptance_run_id='$run_id'::uuid
AND (
status <> 'succeeded'
OR billing_status <> 'settled'
);")
side_effects=$(database_query "
SELECT
(SELECT count(*)
FROM settlement_outbox settlement
WHERE settlement.task_id IN (
SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid
)
AND settlement.status <> 'completed')
+
(SELECT count(*)
FROM gateway_tasks task
WHERE task.acceptance_run_id='$run_id'::uuid
AND NOT EXISTS (
SELECT 1
FROM gateway_task_callback_outbox callback
WHERE callback.task_id=task.id
AND callback.status='delivered'
));")
if [[ $incomplete == 0 && $side_effects == 0 ]]; then
break
fi
sleep 2
done
[[ $incomplete == 0 && $side_effects == 0 ]]
local queue_state duplicates settlements wallet_duplicates missing_billings callbacks connections max_connections
local active_workers stale_workers
queue_state=$(database_query "SELECT count(*) FILTER (WHERE status='queued')||':'||count(*) FILTER (WHERE status='running') FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;")
[[ $queue_state == 0:0 ]]
duplicates=$(database_query "SELECT count(*) FROM (SELECT remote_task_id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND remote_task_id IS NOT NULL GROUP BY remote_task_id HAVING count(*)>1) duplicate;")
[[ $duplicates == 0 ]]
settlements=$(database_query "SELECT count(*) FROM (SELECT task_id,action FROM settlement_outbox WHERE task_id IN (SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid) GROUP BY task_id,action HAVING count(*)>1) duplicate;")
[[ $settlements == 0 ]]
wallet_duplicates=$(database_query "
SELECT count(*)
FROM (
SELECT reference_id,transaction_type
FROM gateway_wallet_transactions
WHERE reference_type='gateway_task'
AND reference_id IN (
SELECT id::text FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid
)
GROUP BY reference_id,transaction_type
HAVING count(*)>1
) duplicate;")
missing_billings=$(database_query "
SELECT count(*)
FROM gateway_tasks task
WHERE task.acceptance_run_id='$run_id'::uuid
AND NOT EXISTS (
SELECT 1
FROM gateway_wallet_transactions transaction
WHERE transaction.reference_type='gateway_task'
AND transaction.reference_id=task.id::text
AND transaction.transaction_type='task_billing'
);")
[[ $wallet_duplicates == 0 && $missing_billings == 0 ]]
callbacks=$(database_query "
SELECT count(*)
FROM gateway_task_callback_outbox
WHERE task_id IN (
SELECT id FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid
)
AND status <> 'delivered';")
[[ $callbacks == 0 ]]
active_workers=$(database_query "
SELECT count(*)
FROM gateway_worker_instances
WHERE status='active'
AND heartbeat_at > now()-interval '30 seconds';")
stale_workers=$(database_query "
SELECT count(*)
FROM gateway_worker_instances
WHERE status='active'
AND (
heartbeat_at <= now()-interval '30 seconds'
OR desired_capacity <> $expected_slots
OR capacity_limit <> $expected_slots
OR allocated_capacity > $expected_slots
);")
[[ $active_workers == 2 && $stale_workers == 0 ]]
connections=$(database_query "SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend';")
max_connections=$(database_query "SELECT setting::int FROM pg_settings WHERE name='max_connections';")
(( connections * 4 < max_connections * 3 ))
local pod metrics desired current global allocated active_instances pool_max acquired idle
local canceled_acquires lease_failures lease_lost
for site in ningbo hongkong; do
pod=$(remote_kubectl get pods -n "$namespace" \
-l "app.kubernetes.io/name=easyai-api,easyai.io/site=$site" \
-o 'jsonpath={.items[0].metadata.name}')
metrics=$(remote_kubectl exec -n "$namespace" "$pod" -- \
wget -qO- http://127.0.0.1:8088/metrics)
pool_max=$(awk '$1=="easyai_gateway_postgres_pool_max_connections" {print $2}' <<<"$metrics")
acquired=$(awk '$1=="easyai_gateway_postgres_pool_acquired_connections" {print $2}' <<<"$metrics")
idle=$(awk '$1=="easyai_gateway_postgres_pool_idle_connections" {print $2}' <<<"$metrics")
[[ ${pool_max%.*} == "$expected_api_pool" ]]
if [[ ${acquired%.*} == "$expected_api_pool" && ${idle%.*} == 0 ]]; then
echo "API connection pool saturated for site $site" >&2
return 1
fi
done
for _ in 1 2 3 4 5 6; do
for site in ningbo hongkong; do
pod=$(remote_kubectl get pods -n "$namespace" \
-l "app.kubernetes.io/name=easyai-worker,easyai.io/site=$site" \
-o 'jsonpath={.items[0].metadata.name}')
metrics=$(remote_kubectl exec -n "$namespace" "$pod" -- wget -qO- http://127.0.0.1:8088/metrics)
desired=$(awk '$1=="easyai_gateway_async_worker_desired_capacity" {print $2}' <<<"$metrics")
current=$(awk '$1=="easyai_gateway_async_worker_capacity" {print $2}' <<<"$metrics")
global=$(awk '$1=="easyai_gateway_worker_global_capacity" {print $2}' <<<"$metrics")
allocated=$(awk '$1=="easyai_gateway_worker_allocated_capacity" {print $2}' <<<"$metrics")
active_instances=$(awk '$1=="easyai_gateway_worker_active_instances" {print $2}' <<<"$metrics")
pool_max=$(awk '$1=="easyai_gateway_postgres_pool_max_connections" {print $2}' <<<"$metrics")
acquired=$(awk '$1=="easyai_gateway_postgres_pool_acquired_connections" {print $2}' <<<"$metrics")
idle=$(awk '$1=="easyai_gateway_postgres_pool_idle_connections" {print $2}' <<<"$metrics")
canceled_acquires=$(awk '$1=="easyai_gateway_postgres_pool_canceled_acquire_total" {print $2}' <<<"$metrics")
lease_failures=$(awk 'index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"failure\"}")==1 {print $2}' <<<"$metrics")
lease_lost=$(awk 'index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"lost\"}")==1 {print $2}' <<<"$metrics")
[[ ${desired%.*} == "$expected_slots" && ${current%.*} == "$expected_slots" ]]
[[ ${global%.*} == "$expected_global" && ${allocated%.*} -le "$expected_slots" ]]
[[ ${active_instances%.*} == 2 ]]
[[ ${pool_max%.*} == "$expected_pool" ]]
[[ ${canceled_acquires%.*} == 0 && ${lease_failures%.*} == 0 && ${lease_lost%.*} == 0 ]]
if [[ ${acquired%.*} == "$expected_pool" && ${idle%.*} == 0 ]]; then
echo "Worker connection pool saturated for site $site" >&2
return 1
fi
done
sleep 5
done
verify_cluster_links
curl -fsS --max-time 10 https://ai.51easyai.com/api/v1/healthz >/dev/null
curl -fsS --max-time 10 https://ai.51easyai.com/api/v1/readyz >/dev/null
}
sample_pressure() {
local output=$1
local stop_file=$2
local database_state node_memory_percent pod_memory_mib pool_state pod metrics
local consecutive_failures=0
echo 'timestamp,queued,running,oldest_wait_seconds,db_connections,db_max_connections,node_memory_percent,pod_memory_mib,pool_acquired_max,pool_max,pool_idle_min,empty_acquire_total,canceled_acquire_total,lease_failure_total,lease_lost_total,active_instances_min,allocated_capacity_max' >"$output"
while [[ ! -f $stop_file ]]; do
if ! database_state=$(database_query "
SELECT
count(*) FILTER (WHERE status='queued')||','||
count(*) FILTER (WHERE status='running')||','||
COALESCE(EXTRACT(EPOCH FROM (now()-(min(created_at) FILTER (WHERE status='queued')))),0)::bigint||','||
(SELECT count(*) FROM pg_stat_activity WHERE backend_type='client backend')||','||
(SELECT setting FROM pg_settings WHERE name='max_connections')
FROM gateway_tasks
WHERE acceptance_run_id='$run_id'::uuid;"); then
database_state=
fi
if ! node_memory_percent=$(remote_kubectl top nodes --no-headers 2>/dev/null |
awk '{value=$5; sub(/%$/, "", value); if (value>max) max=value} END {print max+0}'); then
node_memory_percent=
fi
if ! pod_memory_mib=$(current_max_pod_memory_mib 2>/dev/null); then
pod_memory_mib=
fi
if ! pool_state=$(
for workload in api worker; do
for site in ningbo hongkong; do
pod=$(remote_kubectl get pods -n "$namespace" \
-l "app.kubernetes.io/name=easyai-$workload,easyai.io/site=$site" -o json |
jq -r '[.items[]
| select(any(.status.conditions[]?; .type=="Ready" and .status=="True"))
| .metadata.name][0] // empty') || exit
[[ -n $pod ]] || continue
metrics=$(remote_kubectl exec -n "$namespace" "$pod" -- \
wget -qO- http://127.0.0.1:8088/metrics) || exit
awk -v workload="$workload" '
$1=="easyai_gateway_postgres_pool_acquired_connections" {acquired=$2}
$1=="easyai_gateway_postgres_pool_max_connections" {pool_max=$2}
$1=="easyai_gateway_postgres_pool_idle_connections" {idle=$2}
$1=="easyai_gateway_postgres_pool_empty_acquire_total" {empty=$2}
$1=="easyai_gateway_postgres_pool_canceled_acquire_total" {canceled=$2}
index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"failure\"}")==1 {failure=$2}
index($1,"easyai_gateway_concurrency_lease_renewals_total{outcome=\"lost\"}")==1 {lost=$2}
$1=="easyai_gateway_worker_active_instances" {active=$2}
$1=="easyai_gateway_worker_allocated_capacity" {allocated=$2}
END {
if (workload=="api") {
failure=0
lost=0
active=999999
allocated=0
}
print acquired "," pool_max "," idle "," empty "," canceled "," failure "," lost "," active "," allocated
}
' <<<"$metrics"
done
done |
awk -F',' '
NR==1 {
acquired=$1; pool_max=$2; idle=$3; empty=$4; canceled=$5
failure=$6; lost=$7; active=$8; allocated=$9
next
}
{
if ($1>acquired) acquired=$1
if ($2>pool_max) pool_max=$2
if ($3<idle) idle=$3
empty+=$4
canceled+=$5
failure+=$6
lost+=$7
if ($8<active) active=$8
if ($9>allocated) allocated=$9
}
END {print acquired "," pool_max "," idle "," empty "," canceled "," failure "," lost "," active "," allocated}
'
) 2>/dev/null; then
pool_state=
fi
if [[ -z $database_state || -z $node_memory_percent ||
-z $pod_memory_mib || -z $pool_state ]]; then
((consecutive_failures += 1))
if (( consecutive_failures >= 3 )); then
echo 'pressure sampling failed three consecutive times' >&2
return 1
fi
sleep 2
continue
fi
consecutive_failures=0
printf '%s,%s,%s,%s,%s\n' \
"$(date -u '+%Y-%m-%dT%H:%M:%SZ')" "$database_state" "$node_memory_percent" "$pod_memory_mib" "$pool_state" >>"$output"
sleep 5
done
}
verify_pressure_report() {
local report_path=$1
local expected_slots=$2
awk -F',' '
NR == 1 { next }
{
samples++
if ($14 > 0 || $15 > 0) {
failed=1
}
if ($9 >= $10 && $11 == 0) {
saturated++
} else {
saturated=0
}
if ($4 >= 900 || $6 <= 0 || $5 * 4 >= $6 * 3 || $7 >= 80 || $8 >= 1536 ||
$10 <= 0 || saturated >= 6 || $16 < 1 || $16 > 2 || $17 > slots) {
failed=1
}
}
END {
exit !(samples > 0 && failed == 0)
}
' slots="$expected_slots" "$report_path"
}
run_capacity_round() {
local profile=$1
local repetition=$2
local profile_slug
profile_slug=$(printf '%s' "$profile" | tr '[:upper:]' '[:lower:]')
local prefix=$report_root/$profile_slug-$repetition
local pressure_report=$prefix-pressure.csv
local pressure_stop=$temporary_root/pressure-stop
local pressure_pid status=0 pressure_status=0
local baseline_memory_mib
local expected_slots
case $profile in
P24) expected_slots=24 ;;
P28) expected_slots=28 ;;
P32) expected_slots=32 ;;
esac
baseline_memory_mib=$(current_max_pod_memory_mib)
rm -f -- "$pressure_stop"
sample_pressure "$pressure_report" "$pressure_stop" &
pressure_pid=$!
run_load_profile gemini-baseline "$prefix-gemini-baseline.json" || status=$?
if (( status == 0 )); then run_load_profile gemini-large "$prefix-gemini-large.json" || status=$?; fi
if (( status == 0 )); then run_load_profile gemini-peak "$prefix-gemini-peak.json" || status=$?; fi
if (( status == 0 )); then run_load_profile video-throughput "$prefix-video-throughput.json" || status=$?; fi
if (( status == 0 )); then run_recovery_profile "$prefix-video-recovery.json" || status=$?; fi
touch "$pressure_stop"
wait "$pressure_pid" || pressure_status=$?
(( status == 0 && pressure_status == 0 )) || {
(( status != 0 )) && return "$status"
return "$pressure_status"
}
verify_pressure_report "$pressure_report" "$expected_slots"
verify_runtime_gates
wait_for_memory_recovery "$baseline_memory_mib"
}
finish_and_promote() {
local emulator_report=$report_root/emulator-report.json
local finish_response=$temporary_root/finish.json
local mode_response=$temporary_root/mode.json
remote_kubectl exec -n "$namespace" deployment/easyai-acceptance-emulator -- \
wget -qO- http://127.0.0.1:8090/report >"$emulator_report"
local task_count video_task_count gemini_task_count submissions duplicates callbacks
local gemini_requests forced_tasks forced_requests verified_conversions compact_payloads
task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid;")
video_task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='videos.generations' AND run_mode='acceptance';")
gemini_task_count=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='images.edits' AND run_mode='acceptance';")
forced_tasks=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND request::text LIKE '%acceptance-force-conversion%';")
compact_payloads=$(database_query "SELECT count(*) FROM gateway_tasks WHERE acceptance_run_id='$run_id'::uuid AND kind='images.edits' AND (pg_column_size(request) >= 1048576 OR pg_column_size(result) >= 1048576);")
submissions=$(jq -r '.videoSubmissions' "$emulator_report")
duplicates=$(jq -r '.duplicateCallbacks' "$emulator_report")
callbacks=$(jq -r '.callbackEvents' "$emulator_report")
gemini_requests=$(jq -r '.geminiRequests' "$emulator_report")
forced_requests=$(jq -r '.forcedConversionRequests' "$emulator_report")
verified_conversions=$(jq -r '.verifiedConversions' "$emulator_report")
[[ $submissions == "$video_task_count" && $gemini_requests == "$gemini_task_count" ]]
[[ $forced_requests == "$forced_tasks" && $verified_conversions == "$forced_tasks" ]]
[[ $duplicates == 0 && $compact_payloads == 0 ]]
[[ $callbacks -gt 0 ]]
admin_request GET /api/admin/system/acceptance/traffic-mode '' "$mode_response"
verify_release_cas
local promotion_body
promotion_body=$(jq -cn \
--argjson revision "$(jq -r '.revision' "$mode_response")" \
--arg releaseSha "$release_sha" \
--arg apiDigest "$api_digest" \
--arg workerDigest "$worker_digest" \
'{revision:$revision,releaseSha:$releaseSha,apiImageDigest:$apiDigest,workerImageDigest:$workerDigest}')
jq -n \
--arg runId "$run_id" \
--arg releaseSha "$release_sha" \
--arg apiImageDigest "$api_digest" \
--arg workerImageDigest "$worker_digest" \
--arg stableCapacityProfile "$stable_profile" \
--argjson tasks "$task_count" \
--argjson geminiTasks "$gemini_task_count" \
--argjson videoTasks "$video_task_count" \
--argjson upstreamSubmissions "$submissions" \
--argjson callbackDeliveries "$callbacks" \
'{
runId:$runId,
releaseSha:$releaseSha,
apiImageDigest:$apiImageDigest,
workerImageDigest:$workerImageDigest,
stableCapacityProfile:$stableCapacityProfile,
tasks:$tasks,
geminiTasks:$geminiTasks,
videoTasks:$videoTasks,
upstreamSubmissions:$upstreamSubmissions,
callbackDeliveries:$callbackDeliveries,
promotionRequested:true,
passed:true
}' >"$report_root/summary.json"
local finish_body
finish_body=$(jq -cn \
--argjson tasks "$task_count" \
--argjson videos "$video_task_count" \
--argjson submissions "$submissions" \
--arg profile "$stable_profile" \
'{passed:true,report:{tasks:$tasks,videos:$videos,upstreamSubmissions:$submissions,stableCapacityProfile:$profile,queueFinal:0}}')
admin_request POST "/api/admin/system/acceptance/runs/$run_id/finish" "$finish_body" "$finish_response"
admin_request POST "/api/admin/system/acceptance/runs/$run_id/promote" "$promotion_body" "$finish_response"
[[ $(jq -r '.mode' "$finish_response") == live ]]
}
verify_release_cas
ensure_acceptance_user_group
ensure_acceptance_identity_shards
select_acceptance_models
ensure_acceptance_model_access
deploy_protocol_emulator
create_and_activate_run
wait_for_existing_tasks_to_drain
for profile in P24 P28 P32; do
if ! apply_capacity_profile "$profile"; then
failure_reason="failed to apply $profile"
break
fi
profile_passed=true
for repetition in 1 2 3; do
if ! run_capacity_round "$profile" "$repetition"; then
profile_passed=false
failure_reason="$profile repetition $repetition failed"
break
fi
done
if [[ $profile_passed != true ]]; then
apply_capacity_profile "$stable_profile" || true
break
fi
stable_profile=$profile
done
if [[ -n $failure_reason ]]; then
mark_run_failed "$failure_reason"
echo "production_acceptance=FAIL run_id=$run_id reason=$failure_reason traffic_mode=validation restored_profile=$stable_profile" >&2
exit 1
fi
if ! run_load_profile real-canary "$report_root/real-canary.json"; then
failure_reason='real canary failed'
elif ! verify_runtime_gates; then
failure_reason='post-canary runtime gates failed'
fi
if [[ -n $failure_reason ]]; then
mark_run_failed "$failure_reason"
echo "production_acceptance=FAIL run_id=$run_id reason=$failure_reason traffic_mode=validation restored_profile=$stable_profile" >&2
exit 1
fi
finish_and_promote
remote_kubectl delete deployment easyai-acceptance-emulator -n "$namespace" --ignore-not-found >/dev/null ||
echo 'warning: acceptance emulator Deployment cleanup failed' >&2
remote_kubectl delete service easyai-acceptance-emulator -n "$namespace" --ignore-not-found >/dev/null ||
echo 'warning: acceptance emulator Service cleanup failed' >&2
echo "production_acceptance=PASS run_id=$run_id release=$release_sha stable_profile=$stable_profile traffic_mode=live"