Compare commits
18 Commits
fa8abf22db
...
v1.0.0
| Author | SHA1 | Date | |
|---|---|---|---|
| 24017bcb7f | |||
| 40d8f06588 | |||
| c59e522732 | |||
| 8d45ae6e3b | |||
| 2c4f4b10dc | |||
| 520a9092fe | |||
| 9f970495ee | |||
| 3d9ba3ac3d | |||
| 171b71b7e0 | |||
| 2b399d0838 | |||
| f5f45e7afb | |||
| b54cb8878d | |||
| e336638ca8 | |||
| 62f42ed102 | |||
| ecb21bd218 | |||
| e2771826fd | |||
| dec6fac013 | |||
| c494da553a |
@@ -301,8 +301,11 @@ jobs:
|
||||
# App version for the About screen: the git tag if present, else the short SHA
|
||||
# (the test checkout is shallow/untagged, so this is the SHA here — fine).
|
||||
export APP_VERSION="$(git -C "$GITHUB_WORKSPACE" describe --tags --always 2>/dev/null || echo dev)"
|
||||
docker compose --ansi never build --progress plain
|
||||
docker compose --ansi never up -d --remove-orphans
|
||||
# The telegram-local profile brings the bot + its VPN sidecar; prod runs the
|
||||
# bot on its own host instead (deploy/docker-compose.bot.yml), and the prod
|
||||
# main host omits both. Without the profile they would not start here.
|
||||
docker compose --ansi never --profile telegram-local build --progress plain
|
||||
docker compose --ansi never --profile telegram-local up -d --remove-orphans
|
||||
# The config-only services bind-mount the reseeded config dir. A plain `up -d`
|
||||
# leaves them on the previous bind mount (the dir was rm'd + recreated), so a
|
||||
# changed Caddyfile or Grafana dashboard is ignored — force-recreate them to
|
||||
|
||||
@@ -0,0 +1,266 @@
|
||||
# Manual production rollout. Runs ONLY from master, ONLY on workflow_dispatch with
|
||||
# confirm=deploy (development->master is merged + green first; this is the separate,
|
||||
# deliberate prod step). Visible sequential jobs from most to least significant:
|
||||
# build -> deploy-main -> deploy-bot -> verify
|
||||
# The per-service rolling (postgres->backend->gateway->landing->validator->caddy),
|
||||
# health-gating and auto-rollback live in deploy/prod-deploy.sh on the main host and
|
||||
# show in the deploy-main log. Manual post-deploy rollback is prod-rollback.yaml.
|
||||
# See deploy/README.md (prod runbook).
|
||||
name: prod-deploy
|
||||
run-name: "prod deploy ${{ github.sha }}"
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
confirm:
|
||||
description: 'Type "deploy" to confirm a production rollout from master.'
|
||||
required: true
|
||||
default: ""
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
NO_COLOR: "1"
|
||||
DOCKER_CLI_HINTS: "false"
|
||||
REGISTRY: docker.iliadenisov.ru/developer
|
||||
|
||||
jobs:
|
||||
build:
|
||||
if: ${{ github.ref == 'refs/heads/master' && inputs.confirm == 'deploy' }}
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
outputs:
|
||||
tag: ${{ steps.ver.outputs.tag }}
|
||||
env:
|
||||
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||
VITE_TELEGRAM_BOT_ID: ${{ vars.PROD_VITE_TELEGRAM_BOT_ID }}
|
||||
VITE_TELEGRAM_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||
VITE_TELEGRAM_GAME_CHANNEL_NAME: ${{ vars.PROD_VITE_TELEGRAM_GAME_CHANNEL_NAME }}
|
||||
VITE_GATEWAY_URL: ${{ vars.PROD_VITE_GATEWAY_URL }}
|
||||
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Compute version tag
|
||||
id: ver
|
||||
run: echo "tag=$(git describe --tags --always)" >> "$GITHUB_OUTPUT"
|
||||
- name: Registry login
|
||||
run: echo "$PROD_REGISTRY_PASSWORD" | docker login "${REGISTRY%%/*}" -u "$PROD_REGISTRY_USER" --password-stdin
|
||||
- name: Build and push images
|
||||
working-directory: deploy
|
||||
run: |
|
||||
export TAG="${{ steps.ver.outputs.tag }}" APP_VERSION="${{ steps.ver.outputs.tag }}" SCRABBLE_CONFIG_DIR=.
|
||||
# The four main-stack images via compose (reuses the build args, incl. VERSION);
|
||||
# the bot separately, since it is profiled out of the prod compose.
|
||||
docker compose -f docker-compose.yml -f docker-compose.prod.yml build
|
||||
docker compose -f docker-compose.yml -f docker-compose.prod.yml push backend gateway landing validator
|
||||
docker build -f ../platform/telegram/Dockerfile --target bot --build-arg VERSION="$TAG" -t "$REGISTRY/scrabble-telegram-bot:$TAG" ..
|
||||
docker push "$REGISTRY/scrabble-telegram-bot:$TAG"
|
||||
|
||||
deploy-main:
|
||||
needs: build
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
env:
|
||||
TAG: ${{ needs.build.outputs.tag }}
|
||||
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||
GRAFANA_ADMIN_PASSWORD: ${{ secrets.PROD_GRAFANA_ADMIN_PASSWORD }}
|
||||
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||
PROD_BOTLINK_GATEWAY_CERT: ${{ secrets.PROD_BOTLINK_GATEWAY_CERT }}
|
||||
PROD_BOTLINK_GATEWAY_KEY: ${{ secrets.PROD_BOTLINK_GATEWAY_KEY }}
|
||||
GM_BASICAUTH_USER: ${{ vars.PROD_GM_BASICAUTH_USER }}
|
||||
GRAFANA_ROOT_URL: ${{ vars.PROD_GRAFANA_ROOT_URL }}
|
||||
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||
POSTGRES_DB: ${{ vars.PROD_POSTGRES_DB }}
|
||||
POSTGRES_USER: ${{ vars.PROD_POSTGRES_USER }}
|
||||
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Determine previous tag and migration
|
||||
run: |
|
||||
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||
PREV_TAG="$(ssh_main 'cat /opt/scrabble/DEPLOYED_TAG 2>/dev/null || echo none')"
|
||||
MIGRATION=0
|
||||
if [ "$PREV_TAG" != none ]; then
|
||||
if ! git cat-file -e "$PREV_TAG^{commit}" 2>/dev/null; then
|
||||
MIGRATION=1
|
||||
elif git diff --name-only "$PREV_TAG..$TAG" -- backend/internal/postgres/migrations/ | grep -q .; then
|
||||
MIGRATION=1
|
||||
fi
|
||||
fi
|
||||
{ echo "PREV_TAG=$PREV_TAG"; echo "MIGRATION=$MIGRATION"; } >> "$GITHUB_ENV"
|
||||
echo "prev=$PREV_TAG migration=$MIGRATION"
|
||||
- name: Render main env + certs
|
||||
run: |
|
||||
umask 077
|
||||
mkdir -p stage/certs-main
|
||||
cat > stage/env.sh <<EOF
|
||||
export REGISTRY='$REGISTRY'
|
||||
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||
export POSTGRES_DB='${POSTGRES_DB:-scrabble}'
|
||||
export POSTGRES_USER='${POSTGRES_USER:-scrabble}'
|
||||
export POSTGRES_PASSWORD='$POSTGRES_PASSWORD'
|
||||
export GM_BASICAUTH_USER='${GM_BASICAUTH_USER:-gm}'
|
||||
export GM_BASICAUTH_HASH='$GM_BASICAUTH_HASH'
|
||||
export GRAFANA_ADMIN_PASSWORD='$GRAFANA_ADMIN_PASSWORD'
|
||||
export GRAFANA_ROOT_URL='$GRAFANA_ROOT_URL'
|
||||
export CADDY_SITE_ADDRESS='$CADDY_SITE_ADDRESS'
|
||||
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||
export DICT_VERSION='$DICT_VERSION'
|
||||
export APP_VERSION='$TAG'
|
||||
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||
export GATEWAY_ABUSE_BAN_ENABLED='true'
|
||||
EOF
|
||||
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-main/ca.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_GATEWAY_CERT" > stage/certs-main/gateway.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_GATEWAY_KEY" > stage/certs-main/gateway.key
|
||||
chmod 644 stage/certs-main/*
|
||||
- name: Deploy the main host
|
||||
run: |
|
||||
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||
ssh_main 'mkdir -p /opt/scrabble/compose'
|
||||
tar -C deploy -czf - docker-compose.yml docker-compose.prod.yml prod-deploy.sh \
|
||||
| ssh_main 'tar -C /opt/scrabble/compose -xzf -'
|
||||
tar -C deploy -czf - caddy otelcol prometheus tempo grafana \
|
||||
| ssh_main 'tar -C /opt/scrabble -xzf -'
|
||||
tar -C stage -czf - certs-main \
|
||||
| ssh_main 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.sh "deploy@$MAIN_HOST:/opt/scrabble/env.sh"
|
||||
echo "$PROD_REGISTRY_PASSWORD" | ssh_main "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||
ssh_main "TAG='$TAG' PREV_TAG='$PREV_TAG' MIGRATION='$MIGRATION' bash /opt/scrabble/compose/prod-deploy.sh"
|
||||
|
||||
deploy-bot:
|
||||
needs: [build, deploy-main]
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
env:
|
||||
TAG: ${{ needs.build.outputs.tag }}
|
||||
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
TG_HOST: ${{ vars.PROD_TG_HOST }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||
TELEGRAM_PROMO_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_PROMO_BOT_TOKEN }}
|
||||
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||
PROD_BOTLINK_BOT_CERT: ${{ secrets.PROD_BOTLINK_BOT_CERT }}
|
||||
PROD_BOTLINK_BOT_KEY: ${{ secrets.PROD_BOTLINK_BOT_KEY }}
|
||||
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||
TELEGRAM_GAME_CHANNEL_ID: ${{ vars.PROD_TELEGRAM_GAME_CHANNEL_ID }}
|
||||
TELEGRAM_CHAT_ID: ${{ vars.PROD_TELEGRAM_CHAT_ID }}
|
||||
TELEGRAM_BOT_USERNAME: ${{ vars.PROD_TELEGRAM_BOT_USERNAME }}
|
||||
TELEGRAM_BOT_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Render bot env + certs
|
||||
run: |
|
||||
umask 077
|
||||
mkdir -p stage/certs-bot
|
||||
cat > stage/env.bot.sh <<EOF
|
||||
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||
export BOT_IMAGE='$REGISTRY/scrabble-telegram-bot:$TAG'
|
||||
export BOTLINK_GATEWAY_ADDR='$MAIN_HOST:9443'
|
||||
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||
export TELEGRAM_GAME_CHANNEL_ID='$TELEGRAM_GAME_CHANNEL_ID'
|
||||
export TELEGRAM_CHAT_ID='$TELEGRAM_CHAT_ID'
|
||||
export TELEGRAM_PROMO_BOT_TOKEN='$TELEGRAM_PROMO_BOT_TOKEN'
|
||||
export TELEGRAM_BOT_USERNAME='$TELEGRAM_BOT_USERNAME'
|
||||
export TELEGRAM_BOT_LINK='$TELEGRAM_BOT_LINK'
|
||||
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||
EOF
|
||||
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-bot/ca.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_BOT_CERT" > stage/certs-bot/bot.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_BOT_KEY" > stage/certs-bot/bot.key
|
||||
chmod 644 stage/certs-bot/*
|
||||
- name: Deploy the bot host
|
||||
run: |
|
||||
ssh_tg() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$TG_HOST" "$@"; }
|
||||
ssh_tg 'mkdir -p /opt/scrabble/compose'
|
||||
tar -C deploy -czf - docker-compose.bot.yml | ssh_tg 'tar -C /opt/scrabble/compose -xzf -'
|
||||
tar -C stage -czf - certs-bot \
|
||||
| ssh_tg 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.bot.sh "deploy@$TG_HOST:/opt/scrabble/env.bot.sh"
|
||||
echo "$PROD_REGISTRY_PASSWORD" | ssh_tg "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||
ssh_tg 'set -a; . /opt/scrabble/env.bot.sh; set +a; cd /opt/scrabble/compose;
|
||||
docker compose -f docker-compose.bot.yml pull;
|
||||
docker compose -f docker-compose.bot.yml up -d'
|
||||
ssh_tg 'for i in $(seq 1 20); do
|
||||
s=$(docker inspect -f "{{.State.Status}}" scrabble-telegram-bot 2>/dev/null || echo missing)
|
||||
r=$(docker inspect -f "{{.State.Restarting}}" scrabble-telegram-bot 2>/dev/null || echo true)
|
||||
if [ "$s" = running ] && [ "$r" = false ]; then
|
||||
c1=$(docker inspect -f "{{.RestartCount}}" scrabble-telegram-bot); sleep 5
|
||||
c2=$(docker inspect -f "{{.RestartCount}}" scrabble-telegram-bot)
|
||||
[ "$c1" = "$c2" ] && { echo "bot healthy"; exit 0; }
|
||||
fi
|
||||
sleep 3
|
||||
done
|
||||
echo "bot not healthy:"; docker logs --tail 80 scrabble-telegram-bot; exit 1'
|
||||
|
||||
verify:
|
||||
needs: [deploy-main, deploy-bot]
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
env:
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||
steps:
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Verify the public site
|
||||
run: |
|
||||
domain="${CADDY_SITE_ADDRESS%% *}"
|
||||
ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "for i in \$(seq 1 20); do
|
||||
if curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/ -o /dev/null &&
|
||||
curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/app/ -o /dev/null &&
|
||||
docker run --rm --network scrabble-internal alpine:3.20 wget -q -T 5 -O /dev/null http://backend:8080/readyz; then
|
||||
echo 'public site + /app/ + backend healthy'; exit 0
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
echo 'public verify failed; recent caddy + gateway + backend logs:'
|
||||
docker logs --tail 40 scrabble-caddy; docker logs --tail 40 scrabble-gateway; docker logs --tail 40 scrabble-backend
|
||||
exit 1"
|
||||
@@ -0,0 +1,223 @@
|
||||
# Manual production rollback. Runs ONLY from master, ONLY on workflow_dispatch with
|
||||
# confirm=rollback. Re-deploys an already-published image tag (no build): leave
|
||||
# target_version blank to roll back to the previously deployed version (read from the
|
||||
# main host), or set it to a specific release tag from the Releases page. The
|
||||
# re-deploy is the same rolling, health-gated path as prod-deploy (TAG=target,
|
||||
# MIGRATION=0 — rollback is image-only and never migrates the DB; image rollback is
|
||||
# DB-safe under the expand-contract rule). See deploy/README.md (prod runbook).
|
||||
name: prod-rollback
|
||||
run-name: "prod rollback ${{ inputs.target_version || 'previous' }}"
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
confirm:
|
||||
description: 'Type "rollback" to confirm a production rollback.'
|
||||
required: true
|
||||
default: ""
|
||||
target_version:
|
||||
description: "Release tag to roll back to (blank = the previous deployed version)."
|
||||
required: false
|
||||
default: ""
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
env:
|
||||
NO_COLOR: "1"
|
||||
DOCKER_CLI_HINTS: "false"
|
||||
REGISTRY: docker.iliadenisov.ru/developer
|
||||
|
||||
jobs:
|
||||
rollback-main:
|
||||
if: ${{ github.ref == 'refs/heads/master' && inputs.confirm == 'rollback' }}
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
outputs:
|
||||
target: ${{ steps.resolve.outputs.target }}
|
||||
env:
|
||||
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||
GRAFANA_ADMIN_PASSWORD: ${{ secrets.PROD_GRAFANA_ADMIN_PASSWORD }}
|
||||
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||
PROD_BOTLINK_GATEWAY_CERT: ${{ secrets.PROD_BOTLINK_GATEWAY_CERT }}
|
||||
PROD_BOTLINK_GATEWAY_KEY: ${{ secrets.PROD_BOTLINK_GATEWAY_KEY }}
|
||||
GM_BASICAUTH_USER: ${{ vars.PROD_GM_BASICAUTH_USER }}
|
||||
GRAFANA_ROOT_URL: ${{ vars.PROD_GRAFANA_ROOT_URL }}
|
||||
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||
POSTGRES_DB: ${{ vars.PROD_POSTGRES_DB }}
|
||||
POSTGRES_USER: ${{ vars.PROD_POSTGRES_USER }}
|
||||
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||
INPUT_TARGET: ${{ inputs.target_version }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Resolve rollback target
|
||||
id: resolve
|
||||
run: |
|
||||
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||
CURRENT="$(ssh_main 'cat /opt/scrabble/DEPLOYED_TAG 2>/dev/null || echo none')"
|
||||
if [ -n "$INPUT_TARGET" ]; then
|
||||
TARGET="$INPUT_TARGET"
|
||||
else
|
||||
TARGET="$(ssh_main 'cat /opt/scrabble/PREVIOUS_TAG 2>/dev/null || echo none')"
|
||||
fi
|
||||
if [ -z "$TARGET" ] || [ "$TARGET" = none ]; then
|
||||
echo "no rollback target (no PREVIOUS_TAG on the host and no target_version input)"; exit 1
|
||||
fi
|
||||
if [ "$TARGET" = "$CURRENT" ]; then
|
||||
echo "target $TARGET is already the deployed version; nothing to do"; exit 1
|
||||
fi
|
||||
echo "rolling back: current=$CURRENT -> target=$TARGET"
|
||||
echo "target=$TARGET" >> "$GITHUB_OUTPUT"
|
||||
{ echo "TARGET=$TARGET"; echo "CURRENT=$CURRENT"; } >> "$GITHUB_ENV"
|
||||
- name: Render main env + certs
|
||||
run: |
|
||||
umask 077
|
||||
mkdir -p stage/certs-main
|
||||
cat > stage/env.sh <<EOF
|
||||
export REGISTRY='$REGISTRY'
|
||||
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||
export POSTGRES_DB='${POSTGRES_DB:-scrabble}'
|
||||
export POSTGRES_USER='${POSTGRES_USER:-scrabble}'
|
||||
export POSTGRES_PASSWORD='$POSTGRES_PASSWORD'
|
||||
export GM_BASICAUTH_USER='${GM_BASICAUTH_USER:-gm}'
|
||||
export GM_BASICAUTH_HASH='$GM_BASICAUTH_HASH'
|
||||
export GRAFANA_ADMIN_PASSWORD='$GRAFANA_ADMIN_PASSWORD'
|
||||
export GRAFANA_ROOT_URL='$GRAFANA_ROOT_URL'
|
||||
export CADDY_SITE_ADDRESS='$CADDY_SITE_ADDRESS'
|
||||
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||
export DICT_VERSION='$DICT_VERSION'
|
||||
export APP_VERSION='$TARGET'
|
||||
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||
export GATEWAY_ABUSE_BAN_ENABLED='true'
|
||||
EOF
|
||||
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-main/ca.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_GATEWAY_CERT" > stage/certs-main/gateway.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_GATEWAY_KEY" > stage/certs-main/gateway.key
|
||||
chmod 644 stage/certs-main/*
|
||||
- name: Roll the main host back
|
||||
run: |
|
||||
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||
ssh_main 'mkdir -p /opt/scrabble/compose'
|
||||
tar -C deploy -czf - docker-compose.yml docker-compose.prod.yml prod-deploy.sh \
|
||||
| ssh_main 'tar -C /opt/scrabble/compose -xzf -'
|
||||
tar -C deploy -czf - caddy otelcol prometheus tempo grafana \
|
||||
| ssh_main 'tar -C /opt/scrabble -xzf -'
|
||||
tar -C stage -czf - certs-main \
|
||||
| ssh_main 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.sh "deploy@$MAIN_HOST:/opt/scrabble/env.sh"
|
||||
echo "$PROD_REGISTRY_PASSWORD" | ssh_main "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||
# Image-only rollback: no migration window (TAG=target, MIGRATION=0). A failed
|
||||
# rollback's auto-revert returns to the current version (PREV_TAG=$CURRENT).
|
||||
ssh_main "TAG='$TARGET' PREV_TAG='$CURRENT' MIGRATION=0 bash /opt/scrabble/compose/prod-deploy.sh"
|
||||
|
||||
rollback-bot:
|
||||
needs: rollback-main
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
env:
|
||||
TARGET: ${{ needs.rollback-main.outputs.target }}
|
||||
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
TG_HOST: ${{ vars.PROD_TG_HOST }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||
TELEGRAM_PROMO_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_PROMO_BOT_TOKEN }}
|
||||
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||
PROD_BOTLINK_BOT_CERT: ${{ secrets.PROD_BOTLINK_BOT_CERT }}
|
||||
PROD_BOTLINK_BOT_KEY: ${{ secrets.PROD_BOTLINK_BOT_KEY }}
|
||||
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||
TELEGRAM_GAME_CHANNEL_ID: ${{ vars.PROD_TELEGRAM_GAME_CHANNEL_ID }}
|
||||
TELEGRAM_CHAT_ID: ${{ vars.PROD_TELEGRAM_CHAT_ID }}
|
||||
TELEGRAM_BOT_USERNAME: ${{ vars.PROD_TELEGRAM_BOT_USERNAME }}
|
||||
TELEGRAM_BOT_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Render bot env + certs
|
||||
run: |
|
||||
umask 077
|
||||
mkdir -p stage/certs-bot
|
||||
cat > stage/env.bot.sh <<EOF
|
||||
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||
export BOT_IMAGE='$REGISTRY/scrabble-telegram-bot:$TARGET'
|
||||
export BOTLINK_GATEWAY_ADDR='$MAIN_HOST:9443'
|
||||
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||
export TELEGRAM_GAME_CHANNEL_ID='$TELEGRAM_GAME_CHANNEL_ID'
|
||||
export TELEGRAM_CHAT_ID='$TELEGRAM_CHAT_ID'
|
||||
export TELEGRAM_PROMO_BOT_TOKEN='$TELEGRAM_PROMO_BOT_TOKEN'
|
||||
export TELEGRAM_BOT_USERNAME='$TELEGRAM_BOT_USERNAME'
|
||||
export TELEGRAM_BOT_LINK='$TELEGRAM_BOT_LINK'
|
||||
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||
EOF
|
||||
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-bot/ca.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_BOT_CERT" > stage/certs-bot/bot.crt
|
||||
printf '%s\n' "$PROD_BOTLINK_BOT_KEY" > stage/certs-bot/bot.key
|
||||
chmod 644 stage/certs-bot/*
|
||||
- name: Roll the bot host back
|
||||
run: |
|
||||
ssh_tg() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$TG_HOST" "$@"; }
|
||||
ssh_tg 'mkdir -p /opt/scrabble/compose'
|
||||
tar -C deploy -czf - docker-compose.bot.yml | ssh_tg 'tar -C /opt/scrabble/compose -xzf -'
|
||||
tar -C stage -czf - certs-bot \
|
||||
| ssh_tg 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.bot.sh "deploy@$TG_HOST:/opt/scrabble/env.bot.sh"
|
||||
echo "$PROD_REGISTRY_PASSWORD" | ssh_tg "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||
ssh_tg 'set -a; . /opt/scrabble/env.bot.sh; set +a; cd /opt/scrabble/compose;
|
||||
docker compose -f docker-compose.bot.yml pull;
|
||||
docker compose -f docker-compose.bot.yml up -d'
|
||||
|
||||
verify:
|
||||
needs: [rollback-main, rollback-bot]
|
||||
runs-on: ubuntu-latest
|
||||
defaults:
|
||||
run:
|
||||
shell: bash
|
||||
env:
|
||||
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||
steps:
|
||||
- name: Set up SSH
|
||||
run: |
|
||||
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||
- name: Verify the public site
|
||||
run: |
|
||||
domain="${CADDY_SITE_ADDRESS%% *}"
|
||||
ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "for i in \$(seq 1 20); do
|
||||
if curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/ -o /dev/null &&
|
||||
docker run --rm --network scrabble-internal alpine:3.20 wget -q -T 5 -O /dev/null http://backend:8080/readyz; then
|
||||
echo 'rolled-back site healthy'; exit 0
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
echo 'verify failed'; docker logs --tail 40 scrabble-caddy; docker logs --tail 40 scrabble-backend; exit 1"
|
||||
@@ -51,7 +51,7 @@ independent (see ARCHITECTURE §9.1).
|
||||
| 15 | Dual Telegram bots & language-gated variants | **done** |
|
||||
| 16 | Deploy infra & test contour (Dockerfiles, gateway static UI, compose, observability) | **done** |
|
||||
| 17 | Test-contour verification & defect fixes | **done** |
|
||||
| 18 | Prod contour deploy (SSH export/import, manual after merge) | todo |
|
||||
| 18 | Prod contour deploy (registry, two-host, rolling + auto-rollback; manual after merge) | machinery built; first cutover pending DNS |
|
||||
| 19 | User feedback (in-app submit + attachment, admin review/reply, account roles) | **done** |
|
||||
|
||||
Scaffolding is incremental: `go.work` lists only existing modules; each stage
|
||||
@@ -413,18 +413,28 @@ raw list is kept here as the record of what the first contour run surfaced.
|
||||
"что-то пошло не так". при этом "new -> эрудит" работает. Попробуй посмотреть в логах сейчас, может что-то есть. Или как-то иначе проанализируй, или давай вместе будем смотреть, если не получится.
|
||||
|
||||
### Stage 18 — Prod contour deploy
|
||||
Scope: the **production contour** on a remote host over SSH. Deploy by **container export/import**
|
||||
(`docker save` → `scp`/ssh → `docker load` → `docker compose up` on the remote), the SSH key + host IP
|
||||
in Gitea secrets; **strictly manual** (`workflow_dispatch`) after `development` is merged to `master`
|
||||
(the Stage 16 branch model: `feature/* → development → master`, merge gated green). Two-contour config
|
||||
uses **`TEST_`/`PROD_` secret/variable prefixes** — Gitea 1.26 has no deployment environments (verified:
|
||||
the `environments` API 404s), so a flat prefixed namespace is the convention.
|
||||
Reuses the Stage 16 `deploy/docker-compose.yml` as-is, mapping the **`PROD_`** set onto the same
|
||||
unprefixed compose vars. **No host caddy on prod**, so the contour's own caddy terminates TLS — set
|
||||
`CADDY_SITE_ADDRESS` to the prod domain so caddy does its own ACME (the Caddyfile is already
|
||||
parameterised for this; the test contour leaves it `:80` behind the host caddy).
|
||||
Open details (re-interview): export/import vs a registry trade-off; prod domain/cert source (ACME vs a
|
||||
provided cert) at the contour caddy; prod VPN; rollback.
|
||||
Scope: the **production contour** on **two remote hosts** over SSH — main (full stack, `erudit-game.ru`)
|
||||
and tg (the bot only). Resolved open details (re-interviewed):
|
||||
- **Transport: a registry** (not export/import) — build + push to `docker.iliadenisov.ru`, the hosts pull by tag.
|
||||
- **Cert: ACME** at the contour caddy (`CADDY_SITE_ADDRESS=erudit-game.ru www.erudit-game.ru`, no host caddy).
|
||||
- **No prod VPN** — the bot host has native Bot API egress (verified `api.telegram.org` → 200).
|
||||
- **Rollback** — rolling per-service deploy (least → most dependent), health-gated, auto-rollback to the
|
||||
previous image tag; a maintenance window + consistent `pg_dump` only on a schema migration
|
||||
(expand-contract keeps the auto-rollback image-only; the dump is a manual safety net).
|
||||
|
||||
**Strictly manual** (`workflow_dispatch` from `master`, `confirm=deploy`) after `development → master`
|
||||
is merged green. `TEST_`/`PROD_` prefixed Gitea secrets/variables (Gitea 1.26 has no deployment
|
||||
environments — the `environments` API 404s). Hosts are provisioned by **`deploy/ansible/`** (docker, a
|
||||
non-sudo `deploy` user with the CI key, key-only sshd, ufw, fail2ban). The main host is **launch-sized**
|
||||
(2 vCPU / 1.9 GiB): `docker-compose.prod.yml` trims the R7 limits (`GOMAXPROCS=2`, smaller caps, 7d
|
||||
Prometheus retention) and adds `node_exporter` for host-memory monitoring (launch undersized, resize at
|
||||
Selectel reactively). `vpn`+`bot` are gated to a `telegram-local` compose profile (test only); the prod
|
||||
bot runs standalone from `docker-compose.bot.yml`. `GATEWAY_ABUSE_BAN_ENABLED=true`.
|
||||
|
||||
**Built:** `deploy/ansible/` (both hosts provisioned + verified), the compose split + `node_exporter`,
|
||||
`.gitea/workflows/prod-deploy.yaml` + `deploy/prod-deploy.sh`, the full `PROD_` secret/variable set.
|
||||
**Remaining (acceptance):** the **first live cutover** — waits on the `erudit-game.ru` DNS delegation
|
||||
(`A`/`www` → the main host) that ACME requires; then run the workflow and verify the public site end-to-end.
|
||||
|
||||
### Stage 19 — User feedback *(done)*
|
||||
A user→operator feedback channel, sequenced after the numbered stages but shipped **before** the Stage 18
|
||||
|
||||
+10
-5
@@ -39,8 +39,8 @@ the edge before prod. Each phase maps back to the owner's raw pre-release TODO l
|
||||
| FM | First-move tile draw (official rules): each seated player draws a tile, the one closest to "A" leads (a blank beats every letter), ties re-drawing until a single leader; **honest per-draw `crypto/rand` entropy**, not the bag seed, so the **record** (`game_setup_draws`, migration `00013`) — not a seed — is the only account of the outcome, kept for future **tournaments** (designed as a discrete per-tile "player N draws" step). Friend/AI draws at create; **auto-match draws at *open*** against a synthetic `uuid.Nil` opponent whose draw rows are back-filled on join, so the opener's seat is fixed up front and the existing open-game pre-move is preserved (no reseating, no play-gating). Admin `/_gm/games/:id` gains the recorded draw list + a simple **step-by-step board replay** (`ReplayTimeline`). | owner ad-hoc | **done** |
|
||||
| SB | Single Telegram bot + per-user variant preferences: the two per-language bots collapse into **one** (drop `accounts.service_language`, `supported_languages`, the `*_EN`/`*_RU` env vars and game-language push routing — the single bot renders in the recipient's `preferred_language`); New Game variant gating moves to a profile **`variant_preferences`** set (default Erudit only, Erudit-first, server-enforced on the caller's auto-match/vs-AI/invitation-create paths, an invited friend may accept any variant); env vars collapse to unsuffixed `TELEGRAM_BOT_TOKEN`/`TELEGRAM_GAME_CHANNEL_ID`/`VITE_TELEGRAM_LINK`/`VITE_TELEGRAM_GAME_CHANNEL_NAME` and `GATEWAY_DEFAULT_SUPPORTED_LANGUAGES` is removed; wire drops `service_language`/`supported_languages` (Session, ValidateInitDataResponse) + the push `language` routing field and adds `variant_preferences` to Profile/UpdateProfile. | owner ad-hoc | **done** |
|
||||
| DV | Dictionary version hygiene: CI + image/compose seed track the current release (`v1.2.1`); a **seed-drift guard** records the flat dir's seed in an authoritative `.seed_version` marker so a bumped build seed on a live volume is ignored (it can't relabel live bytes — which would mis-serve the dictionary + void games pinned to the prior label); `DICT_VERSION` is the fresh-volume seed only, a live contour migrates through the admin console | owner ad-hoc | **done** |
|
||||
| TX | Telegram egress off the main host: split the connector into a home **validator** (Mini App / Login-Widget HMAC, no VPN, no Bot API — so game login no longer depends on Telegram being reachable) and a remote **bot** (Bot API long-poll + `sendMessage`) that holds **no inbound port** and dials the gateway over a reverse **mTLS bot-link** (`pkg/proto/botlink/v1`); the gateway funnels out-of-app push (fire-and-forget, at-most-once) and the backend admin broadcasts (a relay that awaits the bot's ack) down the link. The bot is Telegram-rate-limited; **one bot now**, with seams (a bot registry + `owns_updates` + command ids) for N later; **no webhook** (rejected: one URL per token, adds inbound + a static address). The **unified test contour** runs the split (the bot keeps its VPN sidecar and dials the gateway by its internal name; certs from `deploy/gen-certs.sh`). The **prod** wiring — the bot on a separate host (no VPN), the gateway bot-link port published, `PROD_` certs with scheduled rotation, an SSH deploy of both hosts together — is the **deferred final stage** (Stage 18). | owner ad-hoc | **done** (code + test contour; prod wiring → Stage 18) |
|
||||
| AG | Anti-abuse IP ban + honeypot/honeytoken (prod-only): a fail2ban-style in-memory `ratelimit.Banlist` keyed by client IP, fed by sustained rate-limiter rejections (the IP-keyed public/email/admin classes — the user class stays the soft-flag's concern), a **honeypot** decoy path (the contour caddy tags `/.env`, `/.git`, `/wp-*`, … with `X-Scrabble-Honeypot` and routes them to the gateway), and a **honeytoken** (`GATEWAY_HONEYTOKEN`, a planted bearer). The `abuseGuard` edge middleware refuses a banned IP with **429** before any work — closing the R3 gap that the static SPA/landing was outside the token bucket. Off by default — it keys by the real client IP the shared-NAT test contour does not expose (detection still logs there); enabled in prod via `GATEWAY_ABUSE_BAN_ENABLED`. Operators see + lift bans on the console **Throttled** page; the gateway syncs its active set to the backend (`/api/v1/internal/bans/sync`, `internal/banview`) every 30 s and applies operator unbans. | owner ad-hoc | **done** (code + test contour; ban enabled in prod → Stage 18) |
|
||||
| TX | Telegram egress off the main host: split the connector into a home **validator** (Mini App / Login-Widget HMAC, no VPN, no Bot API — so game login no longer depends on Telegram being reachable) and a remote **bot** (Bot API long-poll + `sendMessage`) that holds **no inbound port** and dials the gateway over a reverse **mTLS bot-link** (`pkg/proto/botlink/v1`); the gateway funnels out-of-app push (fire-and-forget, at-most-once) and the backend admin broadcasts (a relay that awaits the bot's ack) down the link. The bot is Telegram-rate-limited; **one bot now**, with seams (a bot registry + `owns_updates` + command ids) for N later; **no webhook** (rejected: one URL per token, adds inbound + a static address). The **unified test contour** runs the split (the bot keeps its VPN sidecar and dials the gateway by its internal name; certs from `deploy/gen-certs.sh`). The **prod** wiring — the bot on a separate host (no VPN), the gateway bot-link port published, `PROD_` certs, an SSH deploy of both hosts together — is **built in Stage 18** (the two-host registry rollout; first cutover pending the `erudit-game.ru` DNS). | owner ad-hoc | **done** (code + test contour; prod wiring built — Stage 18) |
|
||||
| AG | Anti-abuse IP ban + honeypot/honeytoken (prod-only): a fail2ban-style in-memory `ratelimit.Banlist` keyed by client IP, fed by sustained rate-limiter rejections (the IP-keyed public/email/admin classes — the user class stays the soft-flag's concern), a **honeypot** decoy path (the contour caddy tags `/.env`, `/.git`, `/wp-*`, … with `X-Scrabble-Honeypot` and routes them to the gateway), and a **honeytoken** (`GATEWAY_HONEYTOKEN`, a planted bearer). The `abuseGuard` edge middleware refuses a banned IP with **429** before any work — closing the R3 gap that the static SPA/landing was outside the token bucket. Off by default — it keys by the real client IP the shared-NAT test contour does not expose (detection still logs there); enabled in prod via `GATEWAY_ABUSE_BAN_ENABLED`. Operators see + lift bans on the console **Throttled** page; the gateway syncs its active set to the backend (`/api/v1/internal/bans/sync`, `internal/banview`) every 30 s and applies operator unbans. | owner ad-hoc | **done** (code + test contour; ban on in prod via Stage 18 — machinery built, cutover pending DNS) |
|
||||
| CM | Channel-chat moderation + promo bot: a second standalone bot in the bot container answers `/start` with a localized message + a **URL** button into the **main** bot's Mini App (`?startapp`; a `web_app` button would sign initData with the promo token, which the main validator rejects). The **main** bot gates write access in a channel's linked discussion chat. The chat **allows sending by default** and the bot only restricts (Telegram intersects the chat default with the per-user permission, so a per-user grant cannot exceed a deny-by-default group): it **mutes** a member who is not registered or is admin-suspended or holding a new **`chat_muted`** role, and **un-mutes** an eligible one it had muted, for a member currently in the chat (a `getChatMember` guard, since bots cannot list members). Eligibility = `registered AND NOT suspended AND NOT chat_muted` (the game suspension dominates), resolved once in the backend and reached two ways: the bot's `ResolveChatEligibility` on a `chat_member` event over the existing mTLS bot-link, and a backend `chat_access_changed` event → gateway → `ChatGate` command (emitted on block/unblock, a `chat_muted` change, a first registration, or a temporary-block expiry via a sweeper; idempotent). No schema change — `chat_muted` reuses `account_roles`. | owner ad-hoc | **done** |
|
||||
| → | Stage 18 — prod contour deploy | — | see [`PLAN.md`](PLAN.md) |
|
||||
|
||||
@@ -311,7 +311,7 @@ Then Stage 18.
|
||||
hammer (99.97 % rejected, p99 2 ms). **Top finding:** ~14 % `transport_error` on `game.state` at 500
|
||||
players, under CPU saturation (backend/gateway/Postgres each ~1 core) and amplified by the harness's
|
||||
single shared `http2.Transport`; the harness itself peaked at 86 % of a core on the same host, so the
|
||||
figures are pessimistic. Full trip report in [`../loadtest/REPORT-R2.md`](../loadtest/REPORT-R2.md);
|
||||
figures are pessimistic. Full trip report in [`../loadtest/REPORT.md`](../loadtest/REPORT.md);
|
||||
it feeds R3 (h2c `MaxConcurrentStreams`/timeouts, body-size cap), R6 and R7 (per-player transports,
|
||||
separate hardware, pool/limit sizing).
|
||||
- **CI:** `./loadtest/...` added to the path filter + vet/build/test; `go.work.sum` carries the new deps.
|
||||
@@ -454,7 +454,12 @@ Then Stage 18.
|
||||
one connection per player it bursts into its 2-core cap (the residual 2.49 % `transport_error`); backend
|
||||
~0.85 core and postgres ~1.4 cores had headroom; **tempo reached its 1 GiB cap**; the backend pool sat at
|
||||
its `MaxOpenConns=25` cap (28 backends); docker logs were unbounded (~14 MiB / 30 min on the backend at
|
||||
info). Full write-up in [`../loadtest/REPORT-R7.md`](../loadtest/REPORT-R7.md).
|
||||
info). Full write-up in [`../loadtest/REPORT.md`](../loadtest/REPORT.md). *(Superseded in part: a
|
||||
later pass modelling the `game.evaluate` hot path traced the gateway's CPU appetite to
|
||||
**gateway→backend connection churn** — the default 2-idle-connection HTTP transport — not proxying
|
||||
work. Pooling the connections cut peak gateway CPU ~7× (~1.75 → ~0.26 cores at 500 players) and
|
||||
removed the ephemeral-port-exhaustion cliff behind the residual `transport_error`, so the gateway is
|
||||
no longer the binding constraint — postgres is. The 3-core gateway cap below is now generous headroom.)*
|
||||
- **Round-2 tuning (owner-agreed, all in `deploy/docker-compose.yml`, no code change):** gateway **2 → 3
|
||||
cores + `GOMAXPROCS=3`**; tempo memory **1 → 2 GiB**; backend `MAX_OPEN_CONNS` **25 → 40**; a json-file
|
||||
**log-rotation** default (10m × 3) applied contour-wide via a YAML anchor (level stays info).
|
||||
@@ -464,7 +469,7 @@ Then Stage 18.
|
||||
**burst** run (a single 100 → 500 jump) pegged the gateway at 3 cores (≈296 % sustained, 9.27 % error),
|
||||
confirming it is **connection-CPU-bound** — a true arrival spike is a **horizontal-scaling** lever, not
|
||||
more cores per node (recorded in the prod-sizing recommendation).
|
||||
- **No schema change → no contour DB wipe.** Bake-back: `loadtest/REPORT-R7.md` (new), `loadtest/README.md`,
|
||||
- **No schema change → no contour DB wipe.** Bake-back: `loadtest/REPORT.md`, `loadtest/README.md`,
|
||||
`docs/TESTING.md`, the telemetry/observability section of `docs/ARCHITECTURE.md`, the repo-layout line in `CLAUDE.md`.
|
||||
|
||||
- **UI — Tab-bar navigation redesign** (owner ad-hoc, not on the raw TODO list): drop the hamburger
|
||||
|
||||
+3
-1
@@ -33,7 +33,9 @@ COPY backend ./backend
|
||||
# Reduce the workspace to what the backend needs: backend + pkg. loadtest and the
|
||||
# gateway replace it requires are not in this context, so drop both.
|
||||
RUN go work edit -dropuse=./gateway -dropuse=./platform/telegram -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/backend ./backend/cmd/backend
|
||||
# VERSION (the deploy passes the git tag) is stamped into the binary via the linker.
|
||||
ARG VERSION=dev
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/backend ./backend/cmd/backend
|
||||
|
||||
# --- runtime -----------------------------------------------------------------
|
||||
FROM gcr.io/distroless/static-debian12:nonroot
|
||||
|
||||
@@ -63,6 +63,7 @@ type gameCache struct {
|
||||
|
||||
type cachedGame struct {
|
||||
game *engine.Game
|
||||
seats []Seat
|
||||
variant string
|
||||
lastAccess time.Time
|
||||
}
|
||||
@@ -71,24 +72,27 @@ func newGameCache(ttl time.Duration, now func() time.Time) *gameCache {
|
||||
return &gameCache{entries: make(map[uuid.UUID]*cachedGame), ttl: ttl, now: now}
|
||||
}
|
||||
|
||||
// get returns the live game for id and refreshes its idle timer, or (nil, false).
|
||||
func (c *gameCache) get(id uuid.UUID) (*engine.Game, bool) {
|
||||
// get returns the live game and its immutable seat list for id and refreshes its idle
|
||||
// timer, or (nil, nil, false). The seats let a read check membership (and label seats)
|
||||
// without re-loading the game from the store, since seats never change after a game starts.
|
||||
func (c *gameCache) get(id uuid.UUID) (*engine.Game, []Seat, bool) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
e, ok := c.entries[id]
|
||||
if !ok {
|
||||
return nil, false
|
||||
return nil, nil, false
|
||||
}
|
||||
e.lastAccess = c.now()
|
||||
return e.game, true
|
||||
return e.game, e.seats, true
|
||||
}
|
||||
|
||||
// put stores g as the live game for id. variant labels the entry so the active-
|
||||
// games gauge can report counts by variant without inspecting engine internals.
|
||||
func (c *gameCache) put(id uuid.UUID, g *engine.Game, variant string) {
|
||||
// put stores g as the live game for id together with its seat list. variant labels the
|
||||
// entry so the active-games gauge can report counts by variant without inspecting engine
|
||||
// internals; seats are the game's immutable seat standings for the membership fast path.
|
||||
func (c *gameCache) put(id uuid.UUID, g *engine.Game, variant string, seats []Seat) {
|
||||
c.mu.Lock()
|
||||
defer c.mu.Unlock()
|
||||
c.entries[id] = &cachedGame{game: g, variant: variant, lastAccess: c.now()}
|
||||
c.entries[id] = &cachedGame{game: g, seats: seats, variant: variant, lastAccess: c.now()}
|
||||
}
|
||||
|
||||
// remove drops id from the cache (used on a finished game and after a failed
|
||||
|
||||
@@ -94,8 +94,8 @@ func TestGameCacheEviction(t *testing.T) {
|
||||
cur := time.Unix(1_700_000_000, 0)
|
||||
cache := newGameCache(time.Hour, func() time.Time { return cur })
|
||||
id := uuid.New()
|
||||
cache.put(id, nil, "scrabble_en")
|
||||
if _, ok := cache.get(id); !ok {
|
||||
cache.put(id, nil, "scrabble_en", nil)
|
||||
if _, _, ok := cache.get(id); !ok {
|
||||
t.Fatal("game must be resident after put")
|
||||
}
|
||||
cur = cur.Add(30 * time.Minute)
|
||||
@@ -104,7 +104,7 @@ func TestGameCacheEviction(t *testing.T) {
|
||||
if n := cache.sweep(); n != 1 {
|
||||
t.Errorf("sweep evicted %d, want 1", n)
|
||||
}
|
||||
if _, ok := cache.get(id); ok {
|
||||
if _, _, ok := cache.get(id); ok {
|
||||
t.Error("game must be evicted after idle TTL")
|
||||
}
|
||||
if cache.size() != 0 {
|
||||
|
||||
@@ -287,12 +287,12 @@ func (svc *Service) Create(ctx context.Context, params CreateParams) (Game, erro
|
||||
if err := svc.store.CreateGame(ctx, ins, seats, seeding.draws); err != nil {
|
||||
return Game{}, err
|
||||
}
|
||||
svc.cache.put(id, g, params.Variant.String())
|
||||
svc.metrics.recordStarted(ctx, params.Variant, params.VsAI)
|
||||
created, err := svc.store.GetGame(ctx, id)
|
||||
if err != nil {
|
||||
return Game{}, err
|
||||
}
|
||||
svc.cache.put(id, g, params.Variant.String(), created.Seats)
|
||||
// Honest-AI game seated with a robot: if the robot moves first, reply at once
|
||||
// (the periodic driver is the fallback). No-op for every human-only game.
|
||||
svc.triggerAI(created)
|
||||
@@ -890,26 +890,35 @@ func (svc *Service) timeoutGame(ctx context.Context, gameID uuid.UUID, now time.
|
||||
// EvaluatePlay previews a tentative play for a seated player against the current
|
||||
// board without committing it: whether it is legal and what it would score.
|
||||
func (svc *Service) EvaluatePlay(ctx context.Context, gameID, accountID uuid.UUID, tiles []engine.TileRecord) (EvalResult, error) {
|
||||
unlock := svc.locks.lock(gameID)
|
||||
defer unlock()
|
||||
|
||||
// Hot path: an active game stays cached — the engine game is mutated in place across
|
||||
// moves and evicted only when it finishes — so on a hit the cached live game and its
|
||||
// immutable seat list answer the membership check and the score with no DB read. This
|
||||
// preview is fired on every tile placement, the hottest gameplay call at scale.
|
||||
g, seats, ok := svc.cache.get(gameID)
|
||||
if !ok {
|
||||
// Cold path: load and validate from the store, then replay into the cache.
|
||||
pre, err := svc.store.GetGame(ctx, gameID)
|
||||
if err != nil {
|
||||
return EvalResult{}, err
|
||||
}
|
||||
if _, ok := pre.seatOf(accountID); !ok {
|
||||
return EvalResult{}, ErrNotAPlayer
|
||||
}
|
||||
if pre.Status == StatusFinished {
|
||||
return EvalResult{}, ErrFinished
|
||||
}
|
||||
|
||||
unlock := svc.locks.lock(gameID)
|
||||
defer unlock()
|
||||
g, err := svc.liveGame(ctx, pre)
|
||||
if err != nil {
|
||||
if g, err = svc.liveGame(ctx, pre); err != nil {
|
||||
return EvalResult{}, err
|
||||
}
|
||||
seats = pre.Seats
|
||||
}
|
||||
if !seatedIn(seats, accountID) {
|
||||
return EvalResult{}, ErrNotAPlayer
|
||||
}
|
||||
|
||||
validateStart := time.Now()
|
||||
rec, err := g.EvaluatePlay(tiles)
|
||||
svc.metrics.recordValidate(ctx, pre.Variant, validateStart)
|
||||
svc.metrics.recordValidate(ctx, g.Variant(), validateStart)
|
||||
if err != nil {
|
||||
if errors.Is(err, engine.ErrIllegalPlay) {
|
||||
return EvalResult{Valid: false}, nil
|
||||
@@ -1359,7 +1368,7 @@ func (svc *Service) ExportGCG(ctx context.Context, gameID uuid.UUID) (string, er
|
||||
// liveGame returns the live engine.Game for pre, rebuilding it from the journal
|
||||
// on a cache miss. Callers must hold the per-game lock.
|
||||
func (svc *Service) liveGame(ctx context.Context, pre Game) (*engine.Game, error) {
|
||||
if g, ok := svc.cache.get(pre.ID); ok {
|
||||
if g, _, ok := svc.cache.get(pre.ID); ok {
|
||||
return g, nil
|
||||
}
|
||||
g, err := svc.replay(ctx, pre)
|
||||
@@ -1374,7 +1383,7 @@ func (svc *Service) liveGame(ctx context.Context, pre Game) (*engine.Game, error
|
||||
}
|
||||
}
|
||||
if !g.Over() {
|
||||
svc.cache.put(pre.ID, g, pre.Variant.String())
|
||||
svc.cache.put(pre.ID, g, pre.Variant.String(), pre.Seats)
|
||||
}
|
||||
return g, nil
|
||||
}
|
||||
|
||||
@@ -355,27 +355,33 @@ func (s *Store) ExpiredOpen(ctx context.Context, now time.Time) ([]OpenGame, err
|
||||
// GetGame loads the games row joined with its seats (ordered by seat), or
|
||||
// ErrNotFound.
|
||||
func (s *Store) GetGame(ctx context.Context, id uuid.UUID) (Game, error) {
|
||||
gstmt := postgres.SELECT(table.Games.AllColumns).
|
||||
FROM(table.Games).
|
||||
// One round-trip: the game joined with its seats. A LEFT JOIN keeps a (would-be)
|
||||
// seatless game returning the game with no seats, exactly as the prior two-query
|
||||
// version did; ORDER BY seat preserves seat order. The games columns repeat per seat
|
||||
// row — cheap at 2-4 seats, and one round-trip instead of two, which matters because
|
||||
// GetGame is the universal "load the game" step on every game operation.
|
||||
stmt := postgres.SELECT(table.Games.AllColumns, table.GamePlayers.AllColumns).
|
||||
FROM(table.Games.LEFT_JOIN(table.GamePlayers, table.GamePlayers.GameID.EQ(table.Games.GameID))).
|
||||
WHERE(table.Games.GameID.EQ(postgres.UUID(id))).
|
||||
LIMIT(1)
|
||||
var grow model.Games
|
||||
if err := gstmt.QueryContext(ctx, s.db, &grow); err != nil {
|
||||
if errors.Is(err, qrm.ErrNoRows) {
|
||||
return Game{}, ErrNotFound
|
||||
ORDER_BY(table.GamePlayers.Seat.ASC())
|
||||
var rows []struct {
|
||||
model.Games
|
||||
model.GamePlayers
|
||||
}
|
||||
if err := stmt.QueryContext(ctx, s.db, &rows); err != nil {
|
||||
return Game{}, fmt.Errorf("game: get %s: %w", id, err)
|
||||
}
|
||||
|
||||
sstmt := postgres.SELECT(table.GamePlayers.AllColumns).
|
||||
FROM(table.GamePlayers).
|
||||
WHERE(table.GamePlayers.GameID.EQ(postgres.UUID(id))).
|
||||
ORDER_BY(table.GamePlayers.Seat.ASC())
|
||||
var srows []model.GamePlayers
|
||||
if err := sstmt.QueryContext(ctx, s.db, &srows); err != nil {
|
||||
return Game{}, fmt.Errorf("game: get seats %s: %w", id, err)
|
||||
if len(rows) == 0 {
|
||||
return Game{}, ErrNotFound
|
||||
}
|
||||
return projectGame(grow, srows)
|
||||
seats := make([]model.GamePlayers, 0, len(rows))
|
||||
for i := range rows {
|
||||
// Skip the phantom all-NULL seat row a LEFT JOIN yields for a seatless game.
|
||||
if rows[i].GamePlayers.GameID == id {
|
||||
seats = append(seats, rows[i].GamePlayers)
|
||||
}
|
||||
}
|
||||
return projectGame(rows[0].Games, seats)
|
||||
}
|
||||
|
||||
// GetGameVariant reads just a game's variant — a cheap single-column lookup the edge uses
|
||||
|
||||
@@ -184,6 +184,18 @@ func (g Game) seatOf(accountID uuid.UUID) (int, bool) {
|
||||
return 0, false
|
||||
}
|
||||
|
||||
// seatedIn reports whether accountID holds a seat in seats. It backs the read-side
|
||||
// membership check against the cached, immutable seat list, so a hot read can skip
|
||||
// loading the game from the store.
|
||||
func seatedIn(seats []Seat, accountID uuid.UUID) bool {
|
||||
for _, s := range seats {
|
||||
if s.AccountID == accountID {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// MoveResult is the outcome of a committed transition: the decoded move and the
|
||||
// post-move game, plus the actor's own refilled rack and the bag size after the draw
|
||||
// (Rack/BagLen), so the mover renders the next state from the response without a
|
||||
|
||||
@@ -543,6 +543,12 @@ func TestEvaluatePlayPreview(t *testing.T) {
|
||||
if bad.Valid {
|
||||
t.Error("disconnected play must be invalid")
|
||||
}
|
||||
|
||||
// A non-seated account cannot preview: with the game warm in the live cache, the
|
||||
// membership check runs against the cached seat list (the hot path that skips GetGame).
|
||||
if _, err := svc.EvaluatePlay(ctx, g.ID, provisionAccount(t), hint.Tiles); !errors.Is(err, game.ErrNotAPlayer) {
|
||||
t.Errorf("evaluate by a non-player = %v, want ErrNotAPlayer", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestConcurrentSubmitSerialized confirms the per-game lock lets only one of two
|
||||
|
||||
+69
-3
@@ -17,11 +17,12 @@ operational reference for **every environment variable**.
|
||||
| `backend` | built (`backend/Dockerfile`) | Domain service; bakes in the DAWG dictionaries; runs migrations at boot. |
|
||||
| `postgres` | `postgres:17-alpine` | Database (named volume, `pg_isready` healthcheck). |
|
||||
| `validator` | built (`platform/telegram/Dockerfile`, target `validator`) | Telegram HMAC validator (no VPN, no Bot API); internal gRPC at `validator:9091`. Game login depends only on this. |
|
||||
| `vpn` + `bot` | sidecar + built (`platform/telegram/Dockerfile`, target `bot`) | Telegram bot; egresses through the AmneziaWG sidecar; holds no inbound port — dials the gateway bot-link (mTLS) at `gateway:9443`. |
|
||||
| `vpn` + `bot` | sidecar + built (`platform/telegram/Dockerfile`, target `bot`) | Telegram bot, gated to the **`telegram-local`** profile; egresses through the AmneziaWG sidecar and dials the gateway bot-link (mTLS) at `gateway:9443`. The test contour activates the profile; the prod **main** host omits it and runs the bot standalone on its **own host** (`docker-compose.bot.yml`, no VPN — native Bot API egress). |
|
||||
| `otelcol` | `otel/opentelemetry-collector-contrib` | OTLP/gRPC `:4317` → Prometheus scrape (`:9464`) + Tempo. |
|
||||
| `prometheus` | `prom/prometheus` | Metrics, 15d retention. |
|
||||
| `prometheus` | `prom/prometheus` | Metrics, 15d retention (7d in prod). |
|
||||
| `tempo` | `grafana/tempo` | Traces, 72h retention. |
|
||||
| `grafana` | `grafana/grafana` | Dashboards (provisioned), anonymous-admin behind caddy's `/_gm/grafana`. |
|
||||
| `node_exporter` | `quay.io/prometheus/node-exporter` | Host CPU/memory/disk metrics (Prometheus job `node`); the OOM signal on the tight prod main host (2 vCPU / 1.9 GiB). |
|
||||
|
||||
Networking: inter-service traffic is on the private `internal` network
|
||||
(project-scoped DNS); only `caddy` joins the shared external `edge` network so the
|
||||
@@ -59,7 +60,6 @@ compose binds from this directory.
|
||||
| Variable | Gitea kind | Purpose |
|
||||
| --- | --- | --- |
|
||||
| `POSTGRES_PASSWORD` | secret | Postgres password (also embedded in `BACKEND_POSTGRES_DSN`). |
|
||||
| `AWG_CONF` | secret | AmneziaWG config for the VPN sidecar (the bot's only Telegram egress in the test contour). **Must not contain a `DNS=` line** — it hijacks the shared netns's resolv.conf and breaks the bot resolving `otelcol` / `gateway`. Without it, Docker's resolver handles `otelcol`, `gateway` and `api.telegram.org`. |
|
||||
| `GM_BASICAUTH_HASH` | secret | bcrypt hash gating `/_gm` (admin console + Grafana). Generate with `docker run --rm caddy:2-alpine caddy hash-password --plaintext '<pw>'`. |
|
||||
| `TELEGRAM_MINIAPP_URL` | variable | The Mini App URL the bot hands out in deep links / buttons. |
|
||||
|
||||
@@ -67,6 +67,13 @@ compose binds from this directory.
|
||||
secret) and the bot (Bot API). It defaults to empty in compose, but both **fail at
|
||||
boot** when it is empty.
|
||||
|
||||
**Conditionally — `AWG_CONF`** (secret): the AmneziaWG config for the VPN sidecar, needed
|
||||
only when the `telegram-local` profile runs (the test contour and local runs with the
|
||||
bot). It is **not** `:?`-guarded — compose interpolates profiled-out services too, so the
|
||||
prod main host (no VPN) must not require it. It **must not contain a `DNS=` line** — that
|
||||
hijacks the shared netns's resolv.conf and breaks the bot resolving `otelcol` / `gateway`;
|
||||
without it Docker's resolver handles `otelcol`, `gateway` and `api.telegram.org`.
|
||||
|
||||
## Optional variables (with defaults)
|
||||
|
||||
| Variable | Gitea kind | Default | Purpose |
|
||||
@@ -110,6 +117,65 @@ collector's / gateway's internal IP is fine (connected route), but its `AWG_CONF
|
||||
which resolves `otelcol`, `gateway` and `api.telegram.org`. `GATEWAY_ADMIN_*` is
|
||||
intentionally **unset** — caddy owns `/_gm` in the contour.
|
||||
|
||||
## Production rollout
|
||||
|
||||
Prod runs on **two hosts** (main = full stack + ACME on the domain; tg = the bot only,
|
||||
native Bot API, no VPN), one-time provisioned by **[`ansible/`](ansible/)** (docker, a
|
||||
non-sudo `deploy` user holding the CI key, key-only sshd, default-deny ufw, fail2ban).
|
||||
Re-run `ansible/` after a host resize — it is idempotent.
|
||||
|
||||
**To roll out:** merge `development → master` (CI green), then run the **`prod-deploy`**
|
||||
workflow manually (Gitea → Actions → prod-deploy → run from `master`, input
|
||||
`confirm=deploy`). It builds + pushes the images to the registry, ships the
|
||||
compose/config/certs/env over SSH, deploys the main host with `prod-deploy.sh` (rolling,
|
||||
health-gated, **auto-rollback to the previous tag**), then the bot host, then probes the
|
||||
public site. After `master` is green this workflow is the **only** thing that touches
|
||||
prod — nothing auto-deploys there. It runs four visible jobs: **build → deploy-main →
|
||||
deploy-bot → verify** (the per-service rolling shows in the deploy-main log).
|
||||
|
||||
**Versioning.** Each release is a git tag `vX.Y.Z` on `master`; the deploy stamps
|
||||
`git describe --tags` into every image tag, every binary (`-ldflags` → `pkg/version` →
|
||||
the `service.version` telemetry attribute) and the SPA About screen. Tag the release
|
||||
before running the deploy:
|
||||
|
||||
```sh
|
||||
git tag -a v1.0.0 -m v1.0.0 && git push origin v1.0.0
|
||||
```
|
||||
|
||||
**Manual rollback** (any time after a successful deploy). Run the **`prod-rollback`**
|
||||
workflow (Gitea → Actions → prod-rollback, `confirm=rollback`). Leave `target_version`
|
||||
blank to roll back to the previously deployed version (read from the host's
|
||||
`PREVIOUS_TAG`), or set it to a release tag from the **Releases** page. It re-deploys
|
||||
that already-published image rolling + health-gated — no rebuild, no DB migration
|
||||
(image rollback is DB-safe under the expand-contract rule). The registry keeps every
|
||||
release tag, so any prior release is reachable.
|
||||
|
||||
**Migrations** must be **expand-contract** (backward-compatible; goose is forward-only):
|
||||
the automatic rollback is image-only and never restores the DB. A deploy that changes
|
||||
`backend/internal/postgres/migrations/` opens a maintenance window — the backend (sole
|
||||
writer) is stopped for a consistent `pg_dump` into `/opt/scrabble/dumps` before the new
|
||||
backend migrates. **Manual DB restore** (only if a migration was destructive):
|
||||
`docker exec -i scrabble-postgres psql -U scrabble -d scrabble -c 'DROP SCHEMA backend CASCADE'`,
|
||||
then pipe the dump into the same `psql`, and redeploy the matching old tag.
|
||||
|
||||
**bot-link cert rotation:** regenerate (`deploy/gen-certs.sh /tmp/c --force`), reset the
|
||||
five `PROD_BOTLINK_*` secrets from `/tmp/c`, and re-run the workflow — both hosts redeploy
|
||||
together with the fresh CA.
|
||||
|
||||
**Sizing / monitoring:** the main host launches undersized (2 vCPU / 1.9 GiB); the prod
|
||||
overlay trims limits + `GOMAXPROCS=2` + 7d Prometheus retention, and `node_exporter` feeds
|
||||
host memory to Grafana (`/_gm/grafana/`). Watch host memory and resize at Selectel when
|
||||
players arrive.
|
||||
|
||||
**`PROD_` Gitea set** (mirrors `TEST_`, mapped onto the unprefixed names above) — secrets:
|
||||
`PROD_{POSTGRES_PASSWORD, GM_BASICAUTH_HASH, GRAFANA_ADMIN_PASSWORD, TELEGRAM_BOT_TOKEN,
|
||||
TELEGRAM_PROMO_BOT_TOKEN, REGISTRY_PASSWORD, SSH_KEY, SSH_KNOWN_HOSTS, BOTLINK_CA,
|
||||
BOTLINK_GATEWAY_CERT, BOTLINK_GATEWAY_KEY, BOTLINK_BOT_CERT, BOTLINK_BOT_KEY}`; variables:
|
||||
`PROD_{REGISTRY_USER, MAIN_HOST, TG_HOST, CADDY_SITE_ADDRESS, GM_BASICAUTH_USER,
|
||||
GRAFANA_ROOT_URL, LOG_LEVEL, DICT_VERSION, TELEGRAM_MINIAPP_URL, TELEGRAM_GAME_CHANNEL_ID,
|
||||
TELEGRAM_CHAT_ID, TELEGRAM_BOT_USERNAME, VITE_TELEGRAM_BOT_ID, VITE_TELEGRAM_LINK,
|
||||
VITE_TELEGRAM_GAME_CHANNEL_NAME}`.
|
||||
|
||||
## Host-side setup (outside this repo)
|
||||
|
||||
- **`edge` network** must exist on the host (`docker network create edge`).
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
# Prod host provisioning (Stage 18)
|
||||
|
||||
Idempotent Ansible that prepares the two production hosts. It installs Docker, a
|
||||
non-sudo `deploy` service account, SSH hardening, a default-deny firewall,
|
||||
fail2ban, unattended security upgrades and time sync. It does **not** deploy the
|
||||
application — that is `.gitea/workflows/prod-deploy.yaml`'s job, running as the
|
||||
`deploy` account this playbook creates.
|
||||
|
||||
Hosts are referenced by `~/.ssh/config` aliases (`scrabble-main-ops`,
|
||||
`scrabble-tg-ops`), so no IPs or key paths live in the repo.
|
||||
|
||||
## Prerequisites (controller)
|
||||
|
||||
- `ansible` with the bundled collections (`community.general`, `community.docker`,
|
||||
`ansible.posix`).
|
||||
- The two hosts reachable as root via the ssh-config aliases, host keys already
|
||||
accepted into `known_hosts` (`host_key_checking = True`).
|
||||
|
||||
## One-time: the CI deploy key
|
||||
|
||||
The CI prod-deploy workflow logs into the hosts as `deploy` using a dedicated
|
||||
key. Generate it once on the controller, authorize its public half via the
|
||||
playbook, and store its private half **only** in the Gitea `PROD_SSH_KEY` secret:
|
||||
|
||||
```sh
|
||||
ssh-keygen -t ed25519 -N '' -C scrabble-ci-deploy \
|
||||
-f ~/.ssh/scrabble_ci_deploy_ed25519
|
||||
# private half -> Gitea secret PROD_SSH_KEY (set via API); never commit it
|
||||
```
|
||||
|
||||
## Run
|
||||
|
||||
```sh
|
||||
cd deploy/ansible
|
||||
ansible-playbook site.yml
|
||||
```
|
||||
|
||||
The playbook reads the public key from `~/.ssh/scrabble_ci_deploy_ed25519.pub` by
|
||||
default; override with `-e deploy_ci_pubkey_path=/path/to/key.pub`. Re-running is
|
||||
safe (idempotent) and survives a host resize.
|
||||
|
||||
## What each host gets
|
||||
|
||||
- **both** (`common`): docker-ce + compose plugin, `daemon.json` (live-restore,
|
||||
10m×3 log rotation), `deploy` user (docker group, no sudo), key-only sshd,
|
||||
`ufw` default-deny incoming + allow SSH, fail2ban sshd jail, unattended
|
||||
upgrades, chrony, `/opt/scrabble/{config,certs,dumps,images}`.
|
||||
- **main**: `ufw` opens 80/443/9443; the external `edge` docker network.
|
||||
- **tg**: verifies direct `api.telegram.org` egress (the no-VPN assumption).
|
||||
@@ -0,0 +1,11 @@
|
||||
[defaults]
|
||||
inventory = inventory.ini
|
||||
roles_path = roles
|
||||
interpreter_python = /usr/bin/python3
|
||||
host_key_checking = True
|
||||
stdout_callback = yaml
|
||||
deprecation_warnings = False
|
||||
retry_files_enabled = False
|
||||
|
||||
[ssh_connection]
|
||||
pipelining = True
|
||||
@@ -0,0 +1,21 @@
|
||||
---
|
||||
# Service account the CI prod-deploy workflow uses to drive docker on the hosts.
|
||||
# Membership in the docker group is root-equivalent (docker socket access), which
|
||||
# is all the deploy workflow needs; the account is deliberately not given sudo.
|
||||
deploy_user: deploy
|
||||
|
||||
# Public half of the dedicated CI deploy SSH key, read from the controller at run
|
||||
# time. The private half is generated on the controller during provisioning and
|
||||
# stored ONLY in the Gitea PROD_SSH_KEY secret; it is never committed. Override the
|
||||
# path with -e deploy_ci_pubkey_path=/path/to/key.pub if the key lives elsewhere.
|
||||
deploy_ci_pubkey_path: "{{ lookup('env', 'HOME') }}/.ssh/scrabble_ci_deploy_ed25519.pub"
|
||||
deploy_ci_pubkey: "{{ lookup('file', deploy_ci_pubkey_path) }}"
|
||||
|
||||
# Base directory the deploy workflow rsyncs compose files, config, certs and dumps
|
||||
# into. Owned by deploy_user so the workflow needs no elevation.
|
||||
scrabble_base_dir: /opt/scrabble
|
||||
|
||||
# Docker daemon json-file log rotation, mirroring the compose x-logging anchor so
|
||||
# the host's own containers (and any ad-hoc runs) rotate identically.
|
||||
docker_log_max_size: "10m"
|
||||
docker_log_max_file: "3"
|
||||
@@ -0,0 +1,19 @@
|
||||
# Production inventory for Stage 18.
|
||||
#
|
||||
# Hosts resolve through the operator's ~/.ssh/config aliases, so HostName (public
|
||||
# IP), User and IdentityFile live there — no IPs or key paths are committed here.
|
||||
# scrabble-main-ops -> main stack host (public IP, domain erudit-game.ru)
|
||||
# scrabble-tg-ops -> Telegram bot host (direct Bot API egress, no VPN)
|
||||
|
||||
[main]
|
||||
scrabble-main-ops
|
||||
|
||||
[tg]
|
||||
scrabble-tg-ops
|
||||
|
||||
[prod:children]
|
||||
main
|
||||
tg
|
||||
|
||||
[prod:vars]
|
||||
ansible_user=root
|
||||
@@ -0,0 +1,15 @@
|
||||
---
|
||||
- name: restart docker
|
||||
ansible.builtin.service:
|
||||
name: docker
|
||||
state: restarted
|
||||
|
||||
- name: reload sshd
|
||||
ansible.builtin.service:
|
||||
name: ssh
|
||||
state: reloaded
|
||||
|
||||
- name: restart fail2ban
|
||||
ansible.builtin.service:
|
||||
name: fail2ban
|
||||
state: restarted
|
||||
@@ -0,0 +1,167 @@
|
||||
---
|
||||
# Common baseline applied to both prod hosts: Docker engine, a non-sudo deploy
|
||||
# service account, SSH hardening, a default-deny firewall, fail2ban, unattended
|
||||
# security upgrades and time sync. Every task is idempotent.
|
||||
|
||||
- name: Install base packages
|
||||
ansible.builtin.apt:
|
||||
name:
|
||||
- ca-certificates
|
||||
- curl
|
||||
- gnupg
|
||||
- ufw
|
||||
- fail2ban
|
||||
- unattended-upgrades
|
||||
- chrony
|
||||
state: present
|
||||
update_cache: true
|
||||
cache_valid_time: 3600
|
||||
|
||||
# --- Docker engine (official repo; trixie is published upstream) ---------------
|
||||
|
||||
- name: Create apt keyring directory
|
||||
ansible.builtin.file:
|
||||
path: /etc/apt/keyrings
|
||||
state: directory
|
||||
mode: "0755"
|
||||
|
||||
- name: Install Docker apt GPG key
|
||||
ansible.builtin.get_url:
|
||||
url: https://download.docker.com/linux/debian/gpg
|
||||
dest: /etc/apt/keyrings/docker.asc
|
||||
mode: "0644"
|
||||
|
||||
- name: Add Docker apt repository
|
||||
ansible.builtin.apt_repository:
|
||||
repo: >-
|
||||
deb [arch=amd64 signed-by=/etc/apt/keyrings/docker.asc]
|
||||
https://download.docker.com/linux/debian {{ ansible_distribution_release }} stable
|
||||
filename: docker
|
||||
state: present
|
||||
|
||||
- name: Install Docker engine and the compose plugin
|
||||
ansible.builtin.apt:
|
||||
name:
|
||||
- docker-ce
|
||||
- docker-ce-cli
|
||||
- containerd.io
|
||||
- docker-buildx-plugin
|
||||
- docker-compose-plugin
|
||||
state: present
|
||||
update_cache: true
|
||||
|
||||
- name: Configure the Docker daemon (live-restore + log rotation)
|
||||
ansible.builtin.template:
|
||||
src: daemon.json.j2
|
||||
dest: /etc/docker/daemon.json
|
||||
mode: "0644"
|
||||
notify: restart docker
|
||||
|
||||
- name: Enable and start Docker
|
||||
ansible.builtin.service:
|
||||
name: docker
|
||||
enabled: true
|
||||
state: started
|
||||
|
||||
# --- Deploy service account ----------------------------------------------------
|
||||
|
||||
- name: Create the deploy service account
|
||||
ansible.builtin.user:
|
||||
name: "{{ deploy_user }}"
|
||||
groups: docker
|
||||
append: true
|
||||
shell: /bin/bash
|
||||
create_home: true
|
||||
|
||||
- name: Ensure the deploy .ssh directory
|
||||
ansible.builtin.file:
|
||||
path: "/home/{{ deploy_user }}/.ssh"
|
||||
state: directory
|
||||
owner: "{{ deploy_user }}"
|
||||
group: "{{ deploy_user }}"
|
||||
mode: "0700"
|
||||
|
||||
- name: Authorize the CI deploy SSH key (exclusive)
|
||||
ansible.builtin.copy:
|
||||
dest: "/home/{{ deploy_user }}/.ssh/authorized_keys"
|
||||
content: "{{ deploy_ci_pubkey }}\n"
|
||||
owner: "{{ deploy_user }}"
|
||||
group: "{{ deploy_user }}"
|
||||
mode: "0600"
|
||||
|
||||
# --- SSH hardening -------------------------------------------------------------
|
||||
|
||||
- name: Harden sshd (key-only auth)
|
||||
ansible.builtin.template:
|
||||
src: sshd-hardening.conf.j2
|
||||
dest: /etc/ssh/sshd_config.d/10-scrabble-hardening.conf
|
||||
mode: "0644"
|
||||
validate: sshd -t -f %s
|
||||
notify: reload sshd
|
||||
|
||||
# --- Firewall (default deny incoming) ------------------------------------------
|
||||
# SSH is allowed before the policy flips so enabling ufw never locks us out.
|
||||
|
||||
- name: Allow SSH through the firewall
|
||||
community.general.ufw:
|
||||
rule: allow
|
||||
name: OpenSSH
|
||||
|
||||
- name: Default-deny incoming, allow outgoing
|
||||
community.general.ufw:
|
||||
direction: "{{ item.direction }}"
|
||||
policy: "{{ item.policy }}"
|
||||
loop:
|
||||
- { direction: incoming, policy: deny }
|
||||
- { direction: outgoing, policy: allow }
|
||||
|
||||
- name: Enable the firewall
|
||||
community.general.ufw:
|
||||
state: enabled
|
||||
|
||||
# --- fail2ban ------------------------------------------------------------------
|
||||
|
||||
- name: Configure the fail2ban sshd jail
|
||||
ansible.builtin.template:
|
||||
src: jail.local.j2
|
||||
dest: /etc/fail2ban/jail.local
|
||||
mode: "0644"
|
||||
notify: restart fail2ban
|
||||
|
||||
- name: Enable and start fail2ban
|
||||
ansible.builtin.service:
|
||||
name: fail2ban
|
||||
enabled: true
|
||||
state: started
|
||||
|
||||
# --- Unattended security upgrades + time sync ----------------------------------
|
||||
|
||||
- name: Enable unattended upgrades
|
||||
ansible.builtin.copy:
|
||||
dest: /etc/apt/apt.conf.d/20auto-upgrades
|
||||
mode: "0644"
|
||||
content: |
|
||||
APT::Periodic::Update-Package-Lists "1";
|
||||
APT::Periodic::Unattended-Upgrade "1";
|
||||
|
||||
- name: Enable and start chrony
|
||||
ansible.builtin.service:
|
||||
name: chrony
|
||||
enabled: true
|
||||
state: started
|
||||
|
||||
# --- Deploy directories --------------------------------------------------------
|
||||
|
||||
- name: Create the scrabble base directories
|
||||
ansible.builtin.file:
|
||||
path: "{{ scrabble_base_dir }}/{{ item }}"
|
||||
state: directory
|
||||
owner: "{{ deploy_user }}"
|
||||
group: "{{ deploy_user }}"
|
||||
mode: "0750"
|
||||
loop:
|
||||
- ""
|
||||
- config
|
||||
- certs
|
||||
- dumps
|
||||
- images
|
||||
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"live-restore": true,
|
||||
"log-driver": "json-file",
|
||||
"log-opts": {
|
||||
"max-size": "{{ docker_log_max_size }}",
|
||||
"max-file": "{{ docker_log_max_file }}"
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
# Managed by Ansible (deploy/ansible).
|
||||
[DEFAULT]
|
||||
bantime = 1h
|
||||
findtime = 10m
|
||||
maxretry = 5
|
||||
backend = systemd
|
||||
|
||||
[sshd]
|
||||
enabled = true
|
||||
@@ -0,0 +1,6 @@
|
||||
# Managed by Ansible (deploy/ansible). Key-only authentication.
|
||||
# root stays reachable by key (prohibit-password) for provisioning re-runs.
|
||||
PasswordAuthentication no
|
||||
PermitRootLogin prohibit-password
|
||||
PubkeyAuthentication yes
|
||||
KbdInteractiveAuthentication no
|
||||
@@ -0,0 +1,18 @@
|
||||
---
|
||||
# Main stack host: public web + bot-link ports and the external 'edge' network
|
||||
# the compose stack attaches caddy to.
|
||||
|
||||
- name: Open public web and bot-link ports
|
||||
community.general.ufw:
|
||||
rule: allow
|
||||
port: "{{ item }}"
|
||||
proto: tcp
|
||||
loop:
|
||||
- "80" # HTTP (ACME challenge + redirect to HTTPS)
|
||||
- "443" # HTTPS (caddy edge)
|
||||
- "9443" # bot-link mTLS (remote bot dials in; mutual TLS gates access)
|
||||
|
||||
- name: Ensure the external 'edge' docker network exists
|
||||
community.docker.docker_network:
|
||||
name: edge
|
||||
state: present
|
||||
@@ -0,0 +1,19 @@
|
||||
---
|
||||
# Telegram bot host: holds no inbound port beyond SSH (the bot dials out to the
|
||||
# Bot API and into the main host's bot-link). We only verify direct Bot API
|
||||
# egress here, since the "no VPN" decision depends on it.
|
||||
|
||||
- name: Verify direct Telegram Bot API egress (no VPN on this host)
|
||||
ansible.builtin.uri:
|
||||
url: https://api.telegram.org/
|
||||
method: GET
|
||||
status_code: [200, 301, 302, 401, 404] # any HTTP reply proves reachability
|
||||
timeout: 10
|
||||
register: tg_egress
|
||||
failed_when: false
|
||||
|
||||
- name: Report Telegram reachability
|
||||
ansible.builtin.debug:
|
||||
msg: >-
|
||||
api.telegram.org reachable:
|
||||
{{ (tg_egress.status | default(0) | int) > 0 }} (status {{ tg_egress.status | default('none') }})
|
||||
@@ -0,0 +1,31 @@
|
||||
---
|
||||
# Stage 18 host provisioning. Idempotent: safe to re-run after a host resize.
|
||||
# Prepares hosts only (docker, hardening, service account, firewall); the
|
||||
# application is deployed separately by .gitea/workflows/prod-deploy.yaml.
|
||||
|
||||
- name: Common baseline (both hosts)
|
||||
hosts: prod
|
||||
become: true
|
||||
pre_tasks:
|
||||
- name: Require a well-formed CI deploy public key
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- deploy_ci_pubkey | length > 0
|
||||
- deploy_ci_pubkey is search('^(ssh|ecdsa)-')
|
||||
fail_msg: >-
|
||||
deploy_ci_pubkey is empty or malformed. Generate the key first
|
||||
(see deploy/ansible/README.md) or override deploy_ci_pubkey_path.
|
||||
roles:
|
||||
- common
|
||||
|
||||
- name: Main stack host
|
||||
hosts: main
|
||||
become: true
|
||||
roles:
|
||||
- main
|
||||
|
||||
- name: Telegram bot host
|
||||
hosts: tg
|
||||
become: true
|
||||
roles:
|
||||
- tg
|
||||
@@ -0,0 +1,53 @@
|
||||
# Production Telegram bot host descriptor (standalone — NOT an overlay). Run only on
|
||||
# the bot host:
|
||||
# docker compose -f docker-compose.bot.yml up -d
|
||||
#
|
||||
# The bot egresses to the Bot API directly (no VPN sidecar) and dials the main host's
|
||||
# published bot-link :9443 over mTLS. It exports no telemetry — otelcol lives on the
|
||||
# main host and is unreachable from here — so observe it via `docker logs` on this host.
|
||||
# Values come from the prod-deploy workflow (PROD_ secrets/variables); BOT_IMAGE is the
|
||||
# pushed registry tag and BOTLINK_GATEWAY_ADDR is the main host's <ip>:9443.
|
||||
name: scrabble-bot
|
||||
|
||||
services:
|
||||
bot:
|
||||
container_name: scrabble-telegram-bot
|
||||
image: ${BOT_IMAGE:?set BOT_IMAGE to the registry tag}
|
||||
restart: unless-stopped
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: "10m"
|
||||
max-file: "3"
|
||||
environment:
|
||||
TELEGRAM_BOT_TOKEN: ${TELEGRAM_BOT_TOKEN:?set TELEGRAM_BOT_TOKEN}
|
||||
TELEGRAM_GAME_CHANNEL_ID: ${TELEGRAM_GAME_CHANNEL_ID:-}
|
||||
TELEGRAM_CHAT_ID: ${TELEGRAM_CHAT_ID:-}
|
||||
TELEGRAM_PROMO_BOT_TOKEN: ${TELEGRAM_PROMO_BOT_TOKEN:-}
|
||||
TELEGRAM_BOT_USERNAME: ${TELEGRAM_BOT_USERNAME:-}
|
||||
TELEGRAM_BOT_LINK: ${TELEGRAM_BOT_LINK:-}
|
||||
TELEGRAM_MINIAPP_URL: ${TELEGRAM_MINIAPP_URL:?set TELEGRAM_MINIAPP_URL}
|
||||
# Real Bot API in prod (the test contour pins TELEGRAM_TEST_ENV=true instead).
|
||||
TELEGRAM_TEST_ENV: "false"
|
||||
TELEGRAM_API_BASE_URL: ${TELEGRAM_API_BASE_URL:-}
|
||||
TELEGRAM_OWNS_UPDATES: "true"
|
||||
# Dials the main host's published bot-link. ServerName stays `gateway` (the cert
|
||||
# SAN), so TLS validation is independent of the dial address.
|
||||
TELEGRAM_GATEWAY_ADDR: ${BOTLINK_GATEWAY_ADDR:?set BOTLINK_GATEWAY_ADDR (main:9443)}
|
||||
TELEGRAM_BOTLINK_SERVER_NAME: gateway
|
||||
TELEGRAM_BOTLINK_TLS_CERT: /certs/bot.crt
|
||||
TELEGRAM_BOTLINK_TLS_KEY: /certs/bot.key
|
||||
TELEGRAM_BOTLINK_TLS_CA: /certs/ca.crt
|
||||
TELEGRAM_LOG_LEVEL: ${LOG_LEVEL:-info}
|
||||
TELEGRAM_SERVICE_NAME: scrabble-telegram-bot
|
||||
# No telemetry export: otelcol is on the main host, unreachable from here.
|
||||
TELEGRAM_OTEL_TRACES_EXPORTER: none
|
||||
TELEGRAM_OTEL_METRICS_EXPORTER: none
|
||||
GOMAXPROCS: "1"
|
||||
volumes:
|
||||
- ${SCRABBLE_CONFIG_DIR:-.}/certs:/certs:ro
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
cpus: "1.0"
|
||||
memory: 256M
|
||||
@@ -0,0 +1,98 @@
|
||||
# Production main-host overlay, applied on top of docker-compose.yml on the main host:
|
||||
# docker compose -f docker-compose.yml -f docker-compose.prod.yml up -d
|
||||
#
|
||||
# It (1) publishes caddy 80/443 — there is no host caddy in prod, so the contour caddy
|
||||
# owns the edge and does its own ACME on CADDY_SITE_ADDRESS — and the gateway bot-link
|
||||
# :9443 the remote bot dials in over mTLS; and (2) retunes the R7 limits down for the
|
||||
# 2 vCPU / 1.9 GiB host (GOMAXPROCS=2, smaller memory caps, shorter Prometheus
|
||||
# retention). The contour launches deliberately undersized at zero players; the added
|
||||
# node_exporter + Grafana watch host memory so it can be resized at Selectel when
|
||||
# traffic arrives.
|
||||
#
|
||||
# The bot + its VPN sidecar are absent here (the telegram-local profile is not
|
||||
# activated); the prod bot runs on its own host from docker-compose.bot.yml.
|
||||
|
||||
services:
|
||||
caddy:
|
||||
ports:
|
||||
- "80:80"
|
||||
- "443:443"
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 96M
|
||||
|
||||
gateway:
|
||||
# Prod pulls the pushed image by tag instead of building locally; the base
|
||||
# build: section stays dormant because the deploy always pulls first.
|
||||
image: ${REGISTRY:?set REGISTRY}/scrabble-gateway:${TAG:?set TAG}
|
||||
ports:
|
||||
- "9443:9443"
|
||||
environment:
|
||||
# 2 vCPU host: align the Go scheduler with the cgroup quota (R7's 3 needs 3 cores).
|
||||
GOMAXPROCS: "2"
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
cpus: "2.0"
|
||||
memory: 384M
|
||||
|
||||
backend:
|
||||
image: ${REGISTRY:?set REGISTRY}/scrabble-backend:${TAG:?set TAG}
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 384M
|
||||
|
||||
postgres:
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 384M
|
||||
|
||||
validator:
|
||||
image: ${REGISTRY:?set REGISTRY}/scrabble-telegram-validator:${TAG:?set TAG}
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 96M
|
||||
|
||||
landing:
|
||||
image: ${REGISTRY:?set REGISTRY}/scrabble-landing:${TAG:?set TAG}
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 64M
|
||||
|
||||
otelcol:
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 256M
|
||||
|
||||
prometheus:
|
||||
command:
|
||||
- --config.file=/etc/prometheus/prometheus.yml
|
||||
- --storage.tsdb.retention.time=7d
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 256M
|
||||
|
||||
tempo:
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 384M
|
||||
|
||||
grafana:
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 256M
|
||||
|
||||
postgres_exporter:
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 64M
|
||||
@@ -71,6 +71,8 @@ services:
|
||||
# Seed dictionary for a FRESH volume; the per-contour value comes from the
|
||||
# deploy env (Gitea TEST_/PROD_DICT_VERSION). See the volume note below.
|
||||
DICT_VERSION: ${DICT_VERSION:-v1.2.1}
|
||||
# Build version stamped into the binary (git tag; see pkg/version).
|
||||
VERSION: ${APP_VERSION:-dev}
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
depends_on:
|
||||
@@ -132,6 +134,8 @@ services:
|
||||
VITE_TELEGRAM_GAME_CHANNEL_NAME: ${VITE_TELEGRAM_GAME_CHANNEL_NAME:-}
|
||||
VITE_GATEWAY_URL: ${VITE_GATEWAY_URL:-}
|
||||
VITE_APP_VERSION: ${APP_VERSION:-dev}
|
||||
# Go binary version (the SPA's VITE_APP_VERSION is the same git tag).
|
||||
VERSION: ${APP_VERSION:-dev}
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
depends_on: [backend]
|
||||
@@ -218,6 +222,8 @@ services:
|
||||
context: ..
|
||||
dockerfile: platform/telegram/Dockerfile
|
||||
target: validator
|
||||
args:
|
||||
VERSION: ${APP_VERSION:-dev}
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
environment:
|
||||
@@ -240,14 +246,22 @@ services:
|
||||
networks: [internal]
|
||||
|
||||
# --- Telegram bot (egress via the VPN sidecar in test; dials the gateway) ---
|
||||
# vpn + bot are gated to the `telegram-local` profile: the test contour runs them
|
||||
# locally (CI passes --profile telegram-local), the prod main host omits them, and
|
||||
# the prod bot runs on its own host from deploy/docker-compose.bot.yml.
|
||||
vpn:
|
||||
container_name: scrabble-telegram-vpn
|
||||
image: docker.iliadenisov.ru/developer/amneziawg-sidecar:latest
|
||||
profiles: ["telegram-local"]
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
privileged: true
|
||||
environment:
|
||||
AWG_CONF: ${AWG_CONF:?set AWG_CONF}
|
||||
# Required by the vpn sidecar, which is gated to the telegram-local profile.
|
||||
# Compose can't scope a `:?` guard to a profile (interpolation runs for
|
||||
# profiled-out services too) and the prod main host has no VPN, so this is a soft
|
||||
# default; the test contour always supplies TEST_AWG_CONF and the sidecar validates it.
|
||||
AWG_CONF: ${AWG_CONF:-}
|
||||
networks:
|
||||
internal:
|
||||
aliases: [telegram]
|
||||
@@ -255,10 +269,13 @@ services:
|
||||
bot:
|
||||
container_name: scrabble-telegram-bot
|
||||
image: scrabble-telegram-bot:latest
|
||||
profiles: ["telegram-local"]
|
||||
build:
|
||||
context: ..
|
||||
dockerfile: platform/telegram/Dockerfile
|
||||
target: bot
|
||||
args:
|
||||
VERSION: ${APP_VERSION:-dev}
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
depends_on: [vpn]
|
||||
@@ -444,6 +461,26 @@ services:
|
||||
memory: 128M
|
||||
networks: [internal]
|
||||
|
||||
# node_exporter exports host CPU/memory/disk metrics. The prod main host runs a tight
|
||||
# 1.9 GiB budget, so host memory pressure — not just per-container docker_stats — is
|
||||
# what warns before an OOM. Prometheus scrapes it at :9100 (see prometheus.yml).
|
||||
node_exporter:
|
||||
container_name: scrabble-node-exporter
|
||||
image: quay.io/prometheus/node-exporter:v1.8.2
|
||||
restart: unless-stopped
|
||||
logging: *default-logging
|
||||
command:
|
||||
- --path.rootfs=/host
|
||||
- --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host)($|/)
|
||||
pid: host
|
||||
volumes:
|
||||
- /:/host:ro,rslave
|
||||
deploy:
|
||||
resources:
|
||||
limits:
|
||||
memory: 64M
|
||||
networks: [internal]
|
||||
|
||||
networks:
|
||||
internal:
|
||||
name: scrabble-internal
|
||||
|
||||
Executable
+143
@@ -0,0 +1,143 @@
|
||||
#!/usr/bin/env bash
|
||||
# Production main-host deploy driver. Runs ON the main host, invoked over SSH by
|
||||
# .gitea/workflows/prod-deploy.yaml as the deploy user (which must already be
|
||||
# `docker login`ed to the registry). It pulls the images at the new tag and rolls
|
||||
# the stack ONE service at a time in dependency order (least -> most dependent),
|
||||
# health-checking after each; any failure rolls the whole stack back to the
|
||||
# previously deployed tag.
|
||||
#
|
||||
# A schema migration adds a maintenance window: the backend (the only writer) is
|
||||
# stopped so a consistent pg_dump is taken before the new backend migrates forward.
|
||||
# Image rollback alone is safe under the expand-contract migration rule, so the
|
||||
# automatic rollback never touches the database; the dump is kept for a MANUAL
|
||||
# restore if a migration turned out to be destructive (see deploy/prod/README.md).
|
||||
#
|
||||
# Required env (exported by the workflow over SSH):
|
||||
# REGISTRY registry namespace, e.g. docker.iliadenisov.ru/developer
|
||||
# TAG new image tag (the deployed git SHA)
|
||||
# PREV_TAG previously deployed tag, or "none" on the first deploy
|
||||
# MIGRATION "1" when the deploy carries a schema migration, else "0"
|
||||
# Optional: COMPOSE_DIR ENV_FILE DUMP_DIR STATE_FILE POSTGRES_USER POSTGRES_DB
|
||||
set -uo pipefail
|
||||
|
||||
# Runtime compose vars (POSTGRES_*, GM_*, GRAFANA_*, CADDY_*, TELEGRAM_*, REGISTRY,
|
||||
# SCRABBLE_CONFIG_DIR, ...) come from a shell-sourceable env file the workflow writes
|
||||
# with single-quoted values. Exporting them into the process environment lets compose
|
||||
# interpolate ${...} without re-parsing the value — a plain --env-file would mangle the
|
||||
# literal '$' in the bcrypt GM_BASICAUTH_HASH.
|
||||
ENV_FILE="${ENV_FILE:-/opt/scrabble/env.sh}"
|
||||
# shellcheck disable=SC1090
|
||||
[ -f "$ENV_FILE" ] && . "$ENV_FILE"
|
||||
|
||||
REGISTRY="${REGISTRY:?REGISTRY required (env.sh)}"
|
||||
TAG="${TAG:?TAG required}"
|
||||
PREV_TAG="${PREV_TAG:-none}"
|
||||
MIGRATION="${MIGRATION:-0}"
|
||||
COMPOSE_DIR="${COMPOSE_DIR:-/opt/scrabble/compose}"
|
||||
DUMP_DIR="${DUMP_DIR:-/opt/scrabble/dumps}"
|
||||
STATE_FILE="${STATE_FILE:-/opt/scrabble/DEPLOYED_TAG}"
|
||||
# The prior deployed tag, preserved on every successful deploy so prod-rollback can
|
||||
# target "the previous version" with no operator input.
|
||||
PREV_STATE_FILE="${PREV_STATE_FILE:-/opt/scrabble/PREVIOUS_TAG}"
|
||||
PG_USER="${POSTGRES_USER:-scrabble}"
|
||||
PG_DB="${POSTGRES_DB:-scrabble}"
|
||||
|
||||
cd "$COMPOSE_DIR" || { echo "compose dir $COMPOSE_DIR missing"; exit 1; }
|
||||
export REGISTRY
|
||||
# otelcol joins the host docker group to read the socket; the GID varies per host.
|
||||
DOCKER_GID="$(getent group docker | cut -d: -f3)"
|
||||
export DOCKER_GID
|
||||
|
||||
dc() { docker compose -f docker-compose.yml -f docker-compose.prod.yml "$@"; }
|
||||
use_tag() { export TAG="$1"; }
|
||||
|
||||
# --- health probes (one-off containers on the contour networks, like CI) --------
|
||||
_probe() { docker run --rm --network "$1" alpine:3.20 wget -q -T 5 -O /dev/null "$2"; }
|
||||
health_backend() { for _ in $(seq 1 20); do _probe scrabble-internal http://backend:8080/readyz && return 0; sleep 3; done; return 1; }
|
||||
health_landing() { for _ in $(seq 1 20); do _probe scrabble-internal http://landing:80/ && return 0; sleep 3; done; return 1; }
|
||||
health_postgres() { for _ in $(seq 1 30); do [ "$(docker inspect -f '{{.State.Health.Status}}' scrabble-postgres 2>/dev/null)" = healthy ] && return 0; sleep 2; done; return 1; }
|
||||
health_running() { # health_running <container>: running, not restarting, stable restart count
|
||||
local n="$1" s r c1 c2
|
||||
for _ in $(seq 1 20); do
|
||||
s="$(docker inspect -f '{{.State.Status}}' "$n" 2>/dev/null || echo missing)"
|
||||
r="$(docker inspect -f '{{.State.Restarting}}' "$n" 2>/dev/null || echo true)"
|
||||
if [ "$s" = running ] && [ "$r" = false ]; then
|
||||
c1="$(docker inspect -f '{{.RestartCount}}' "$n")"; sleep 5
|
||||
c2="$(docker inspect -f '{{.RestartCount}}' "$n")"
|
||||
[ "$c1" = "$c2" ] && return 0
|
||||
fi
|
||||
sleep 3
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
roll() { # roll <service> <health-cmd...>
|
||||
local svc="$1"; shift
|
||||
echo ">>> rolling $svc -> $TAG"
|
||||
dc up -d --no-build --no-deps "$svc" || return 1
|
||||
"$@" || { echo "!!! $svc failed health check"; return 1; }
|
||||
echo "<<< $svc healthy"
|
||||
}
|
||||
|
||||
rollback() {
|
||||
echo "########## ROLLBACK -> $PREV_TAG ##########"
|
||||
if [ "$PREV_TAG" = none ]; then
|
||||
echo "no previous tag (first deploy): cannot roll back; leaving the stack up for inspection."
|
||||
return
|
||||
fi
|
||||
use_tag "$PREV_TAG"
|
||||
dc up -d --no-build --remove-orphans
|
||||
echo "rolled back to $PREV_TAG."
|
||||
[ "$MIGRATION" = 1 ] && echo "NOTE: the DB is forward-migrated; a pre-deploy dump is in $DUMP_DIR — restore manually ONLY if the migration was destructive (see deploy/README.md, prod runbook)."
|
||||
}
|
||||
|
||||
commit_tag() {
|
||||
# Record the just-deployed tag as current, preserving the prior one as previous.
|
||||
[ -f "$STATE_FILE" ] && cp "$STATE_FILE" "$PREV_STATE_FILE"
|
||||
echo "$TAG" > "$STATE_FILE"
|
||||
}
|
||||
|
||||
mkdir -p "$DUMP_DIR"
|
||||
echo "=== prod deploy: tag=$TAG prev=$PREV_TAG migration=$MIGRATION ==="
|
||||
use_tag "$TAG"
|
||||
dc pull
|
||||
|
||||
# First deploy: nothing to roll from; bring the whole stack up and gate on health.
|
||||
if [ -z "$(docker ps -aq -f name=scrabble-backend)" ]; then
|
||||
echo "first deploy: bringing the whole stack up"
|
||||
dc up -d --no-build --remove-orphans || { echo "compose up failed"; exit 1; }
|
||||
health_backend || { echo "backend not ready"; exit 1; }
|
||||
health_landing || { echo "landing not ready"; exit 1; }
|
||||
commit_tag
|
||||
echo "first deploy healthy ($TAG)."
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Migration deploy: freeze writes and snapshot a consistent dump before migrating.
|
||||
if [ "$MIGRATION" = 1 ]; then
|
||||
echo "migration deploy: opening maintenance window (stopping the backend = the only writer)"
|
||||
dc stop backend
|
||||
dump="$DUMP_DIR/pre-$TAG-$(date +%Y%m%d-%H%M%S).sql"
|
||||
if ! docker exec scrabble-postgres pg_dump -U "$PG_USER" -d "$PG_DB" -n backend > "$dump"; then
|
||||
echo "pg_dump failed; restarting the old backend and aborting"
|
||||
dc start backend
|
||||
exit 1
|
||||
fi
|
||||
echo "consistent dump: $dump"
|
||||
fi
|
||||
|
||||
# Roll one service at a time, least -> most dependent; any failure rolls everything back.
|
||||
roll postgres health_postgres || { rollback; exit 1; }
|
||||
roll backend health_backend || { rollback; exit 1; }
|
||||
roll gateway health_running scrabble-gateway || { rollback; exit 1; }
|
||||
roll landing health_landing || { rollback; exit 1; }
|
||||
roll validator health_running scrabble-telegram-validator || { rollback; exit 1; }
|
||||
roll caddy health_running scrabble-caddy || { rollback; exit 1; }
|
||||
|
||||
# Observability + node_exporter: bring up the remainder and pick up any config changes.
|
||||
dc up -d --no-build --remove-orphans || { rollback; exit 1; }
|
||||
|
||||
# Final internal sanity before committing the new tag.
|
||||
health_backend || { rollback; exit 1; }
|
||||
commit_tag
|
||||
echo "=== deploy healthy ($TAG) ==="
|
||||
@@ -18,3 +18,8 @@ scrape_configs:
|
||||
- job_name: postgres_exporter
|
||||
static_configs:
|
||||
- targets: ["postgres_exporter:9187"]
|
||||
# Host-level metrics (memory/CPU/disk). Matters most on the prod main host's tight
|
||||
# 1.9 GiB budget, where total host memory is the OOM-proximity signal.
|
||||
- job_name: node
|
||||
static_configs:
|
||||
- targets: ["node_exporter:9100"]
|
||||
|
||||
+39
-13
@@ -128,7 +128,11 @@ dropped). Horizontal scaling is explicit future work.
|
||||
and GCG are unaffected** (they stay decoded concrete characters, §9.1).
|
||||
- **gateway ↔ backend (sync)**: plain HTTP REST/JSON. The gateway injects
|
||||
`X-User-ID` for authenticated requests; `backend` never re-derives identity
|
||||
from the body.
|
||||
from the body. Because every sync call targets the one backend host, the
|
||||
gateway's REST client widens its keep-alive pool well past the stdlib default
|
||||
of 2 idle connections per host; otherwise the per-request connection churn
|
||||
exhausts ephemeral ports and burns gateway CPU under load (see
|
||||
[`../loadtest/REPORT.md`](../loadtest/REPORT.md)).
|
||||
- **backend → gateway (live)**: a single gRPC server-stream carries live events
|
||||
(your-turn, opponent-moved, chat, nudge). The gateway bridges them to the
|
||||
client's in-app stream while the app is open. Out-of-app delivery uses
|
||||
@@ -1052,8 +1056,10 @@ plaintext relay (`GATEWAY_BOTLINK_RELAY_ADDR`) the backend admin console calls.
|
||||
|
||||
The full contour (`deploy/docker-compose.yml`) runs one `gateway`, one `backend`,
|
||||
one Postgres, the static `landing`, the Telegram `validator` and `bot` (+ the bot's VPN
|
||||
sidecar) and the **observability stack** —
|
||||
OTel Collector (OTLP/gRPC ingest → Prometheus metrics + Tempo traces) and Grafana
|
||||
sidecar — the `bot`+`vpn` pair is gated to a `telegram-local` compose profile so the prod
|
||||
main host can omit them) and the **observability stack** —
|
||||
OTel Collector (OTLP/gRPC ingest → Prometheus metrics + Tempo traces), a `node_exporter`
|
||||
for host CPU/memory (the prod main host's OOM signal), and Grafana
|
||||
with provisioned datasources and dashboards. All services export OTLP to the
|
||||
collector; the bot shares the VPN sidecar's netns, so its `AWG_CONF` must not
|
||||
carry a `DNS=` directive (that would hijack resolv.conf and stop it resolving
|
||||
@@ -1077,16 +1083,36 @@ Two contours, two secret/variable prefixes (`TEST_` / `PROD_`):
|
||||
generated by `deploy/gen-certs.sh` before `compose up`; the bot keeps its VPN sidecar
|
||||
for Telegram egress and dials the gateway by its internal name, so the bot-link stays
|
||||
on the internal network.
|
||||
- **Prod**: a manual SSH deploy after `development → master`. There is no
|
||||
host caddy, so the contour ships its own caddy terminating TLS — set
|
||||
`CADDY_SITE_ADDRESS` to the domain and the caddy does its own ACME. The **bot runs
|
||||
on a separate host** with native Telegram access (no VPN), deployed by SSH alongside
|
||||
the main app (rolled together so the bot-link protocol versions never skew); the
|
||||
gateway **publishes** the bot-link port and the certificates come from `PROD_`
|
||||
secrets — a long-lived CA with leaves rotated by a scheduled job. The bot dials the
|
||||
gateway's public bot-link endpoint and holds no inbound port; login is unaffected if
|
||||
that host or the link is down. *(This prod wiring is the deferred final stage; the
|
||||
code and the unified test contour land first — see `PRERELEASE.md`.)*
|
||||
- **Prod**: a **manual** rollout — `.gitea/workflows/prod-deploy.yaml`, `workflow_dispatch`
|
||||
only (from `master`, `confirm=deploy`), run after `development → master` is merged green.
|
||||
It builds and pushes the images to the registry (`docker.iliadenisov.ru`), then deploys
|
||||
over SSH onto **two hosts** provisioned by `deploy/ansible/` (docker, a non-sudo `deploy`
|
||||
service account holding a dedicated CI key, key-only sshd, default-deny ufw, fail2ban):
|
||||
the **main host** runs the full stack (`docker-compose.yml` + `docker-compose.prod.yml`),
|
||||
the **bot host** runs only the bot (`docker-compose.bot.yml`, no VPN — native Bot API
|
||||
egress, telemetry off). There is no host caddy, so the contour caddy terminates TLS —
|
||||
`CADDY_SITE_ADDRESS` is the domain and caddy does its own ACME. The gateway **publishes**
|
||||
the bot-link `:9443`; the remote bot dials it over mTLS (certs from `PROD_BOTLINK_*`,
|
||||
ServerName `gateway`, so TLS validation is independent of the public dial address), holds
|
||||
no inbound port, and login is unaffected if that host or the link is down.
|
||||
`deploy/prod-deploy.sh` rolls the main stack **one service at a time in dependency order**
|
||||
(postgres → backend → gateway → landing → validator → caddy), health-checking after each;
|
||||
any failure **rolls the whole stack back to the previous image tag**. A **schema migration**
|
||||
adds a maintenance window: the backend (the sole writer) is stopped for a consistent
|
||||
`pg_dump` before the new backend migrates forward — image rollback stays DB-safe under the
|
||||
expand-contract migration rule, and the dump is kept for a manual restore. The workflow runs
|
||||
four visible jobs (build → deploy-main → deploy-bot → verify). Releases are git tags
|
||||
`vX.Y.Z`; the version is stamped into the image tag, every binary (`-ldflags` → `pkg/version`
|
||||
→ the `service.version` telemetry attribute) and the SPA About screen. A separate manual
|
||||
**`prod-rollback`** workflow re-deploys any prior release tag (blank input = the previous
|
||||
deployed version, tracked on the host) over the same rolling, health-gated path — image-only,
|
||||
no DB migration. The main host is
|
||||
intentionally **launch-sized** (2 vCPU / 1.9 GiB): the prod overlay trims the R7 limits
|
||||
(`GOMAXPROCS=2`, smaller caps, 7d Prometheus retention) and a **node_exporter** feeds
|
||||
host-memory metrics to Grafana so it can be resized reactively as players arrive.
|
||||
`GATEWAY_ABUSE_BAN_ENABLED=true` in prod (the per-IP ban is meaningful only with real
|
||||
client IPs). The `vpn`+`bot` pair is gated to a `telegram-local` compose profile the test
|
||||
contour activates; the prod main host omits it.
|
||||
|
||||
## 14. CI & branches
|
||||
|
||||
|
||||
+8
-1
@@ -31,7 +31,10 @@ ephemeral guest. The gateway validates the credential once and mints a thin
|
||||
session token; the backend resolves it to an internal `user_id`. A **Telegram Mini
|
||||
App** launch authenticates from the platform's signed `initData`, themes the UI to
|
||||
the Telegram colours, and — on first contact — seeds the new account's interface
|
||||
language from the Telegram client. Telegram runs a **single bot**: every player uses
|
||||
language from the Telegram client. If a launch cannot reach the backend (for example during a
|
||||
deployment), the Mini App retries quietly and then shows a small "couldn't load" screen with a
|
||||
**Retry** button, rather than dropping to the web sign-in, which has no place inside Telegram.
|
||||
Telegram runs a **single bot**: every player uses
|
||||
the same bot, and all of its chat and out-of-app notifications are written in the
|
||||
player's own **interface language** (en/ru). A separate optional **promo bot** can run alongside the
|
||||
main one — its only job is to answer `/start` with a short message and a button that opens the
|
||||
@@ -56,6 +59,10 @@ reconnect), and pending reads resume on their own — the interface stays usable
|
||||
flashing a red banner each time.
|
||||
|
||||
### Accounts, linking & merge
|
||||
_Sign-in is currently provider-only, so the in-profile linking UI is temporarily hidden; it
|
||||
returns once the anonymous `/app/` guest (whose upgrade path this is) ships. The flow below
|
||||
describes it for when it does._
|
||||
|
||||
First platform contact auto-provisions a durable account. From the profile a player
|
||||
links an email (via a confirm code) or their Telegram (via the web sign-in); a guest
|
||||
who links their first identity becomes a durable account. The "already taken" status
|
||||
|
||||
@@ -32,7 +32,10 @@ top-1 подсказку, безлимитную проверку слова с
|
||||
session-токен; backend сопоставляет его с внутренним `user_id`. Запуск **Telegram
|
||||
Mini App** авторизует по подписанным `initData` платформы, перекрашивает интерфейс
|
||||
в цвета Telegram и — при первом контакте — задаёт язык интерфейса нового аккаунта по
|
||||
языку Telegram-клиента. Telegram держит **единого бота**: все игроки пользуются одним
|
||||
языку Telegram-клиента. Если запуск не может достучаться до бэкенда (например, во время
|
||||
деплоя), Mini App тихо повторяет попытки, а затем показывает небольшой экран «не удалось
|
||||
загрузить» с кнопкой **Повторить**, вместо того чтобы сбрасывать на веб-вход, которому внутри
|
||||
Telegram не место. Telegram держит **единого бота**: все игроки пользуются одним
|
||||
и тем же ботом, а весь его чат и внеприложенческие уведомления пишутся на **языке
|
||||
интерфейса** самого игрока (en/ru). Рядом с основным может работать отдельный опциональный
|
||||
**промо-бот** — его единственная задача отвечать на `/start` коротким сообщением и кнопкой,
|
||||
@@ -57,6 +60,10 @@ Mini App** авторизует по подписанным `initData` плат
|
||||
рабочим вместо красного баннера каждый раз.
|
||||
|
||||
### Аккаунты, привязка и слияние
|
||||
_Вход сейчас только через провайдера, поэтому UI привязки в профиле временно скрыт; он
|
||||
вернётся, когда появится анонимный `/app/`-гость (для апгрейда которого он и нужен). Описание
|
||||
ниже — на этот случай._
|
||||
|
||||
Первый контакт с платформы заводит постоянный аккаунт. Из профиля игрок
|
||||
привязывает email (по confirm-коду) или свой Telegram (через веб-вход); гость,
|
||||
привязавший первую личность, становится постоянным аккаунтом. Факт «личность уже
|
||||
|
||||
+3
-3
@@ -133,9 +133,9 @@ tests or touching CI.
|
||||
engine tests do). It is **not** part of the per-PR suite's behavioural assertions: it
|
||||
runs ad hoc as a one-shot container against the contour, producing a trip report (bugs
|
||||
+ a per-container resource profile) read off the **otelcol `docker_stats` +
|
||||
postgres_exporter** Grafana dashboard on the contour. Two passes are recorded — the
|
||||
early [`REPORT-R2.md`](../loadtest/REPORT-R2.md) and the final, tuned
|
||||
[`REPORT-R7.md`](../loadtest/REPORT-R7.md). See [`../loadtest/README.md`](../loadtest/README.md).
|
||||
postgres_exporter** Grafana dashboard on the contour. The findings — including the
|
||||
`game.evaluate` hot-path model and the gateway→backend connection-pool fix — are written
|
||||
up in [`REPORT.md`](../loadtest/REPORT.md). See [`../loadtest/README.md`](../loadtest/README.md).
|
||||
- **User feedback** — `internal/feedback` unit tests cover the attachment allow-list /
|
||||
content-type and the channel normaliser; the UI covers `detectChannel`, the attachment gate and
|
||||
the feedback wire round-trip (`channel` / `feedback` / `codec` tests) plus a Playwright e2e
|
||||
|
||||
+3
-1
@@ -70,7 +70,9 @@ RUN rm gateway/internal/webui/dist/landing.html
|
||||
# Reduce the workspace to what the gateway needs: gateway + pkg (loadtest is not in
|
||||
# this context; its scrabble/gateway replace targets ./gateway, which is present here).
|
||||
RUN go work edit -dropuse=./backend -dropuse=./platform/telegram -dropuse=./loadtest
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/gateway ./gateway/cmd/gateway
|
||||
# VERSION (the deploy passes the git tag) is stamped into the binary via the linker.
|
||||
ARG VERSION=dev
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/gateway ./gateway/cmd/gateway
|
||||
|
||||
# --- runtime -----------------------------------------------------------------
|
||||
FROM gcr.io/distroless/static-debian12:nonroot AS gateway
|
||||
|
||||
@@ -22,6 +22,19 @@ import (
|
||||
pushv1 "scrabble/pkg/proto/push/v1"
|
||||
)
|
||||
|
||||
// backendMaxIdleConns sizes the REST keep-alive pool to the single backend host. The
|
||||
// default transport caps idle connections per host at 2 (http.DefaultMaxIdleConnsPerHost),
|
||||
// which — since every synchronous client call proxies to that one host — forces a fresh
|
||||
// TCP connection (and a lingering TIME_WAIT socket) for almost every request under load.
|
||||
// That connection churn burns gateway CPU and exhausts ephemeral ports at scale, all
|
||||
// while the backend itself sits near-idle. Pooling the connections lets them be reused.
|
||||
//
|
||||
// The stress harness measured the effect at 500 concurrent players: the churn collapsed
|
||||
// from ~26 500 TIME_WAIT sockets to ~0 and peak gateway CPU from ~1.75 to ~0.26 cores,
|
||||
// with the pool settling at ~225 live connections. 512 keeps ~2x headroom over that
|
||||
// observed peak so a burst never re-caps the pool. See loadtest/REPORT.md.
|
||||
const backendMaxIdleConns = 512
|
||||
|
||||
// Client calls the backend's REST API and opens its push gRPC stream.
|
||||
type Client struct {
|
||||
baseURL string
|
||||
@@ -41,9 +54,14 @@ func New(httpURL, grpcAddr string, timeout time.Duration) (*Client, error) {
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("backendclient: dial push %s: %w", grpcAddr, err)
|
||||
}
|
||||
// Clone the default transport (keeping its proxy, dialer and timeouts) and widen the
|
||||
// idle pool so REST calls to the backend reuse connections instead of churning them.
|
||||
transport := http.DefaultTransport.(*http.Transport).Clone()
|
||||
transport.MaxIdleConns = backendMaxIdleConns
|
||||
transport.MaxIdleConnsPerHost = backendMaxIdleConns
|
||||
return &Client{
|
||||
baseURL: strings.TrimRight(httpURL, "/"),
|
||||
http: &http.Client{Timeout: timeout},
|
||||
http: &http.Client{Timeout: timeout, Transport: transport},
|
||||
conn: conn,
|
||||
push: pushv1.NewPushClient(conn),
|
||||
}, nil
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
package backendclient
|
||||
|
||||
import (
|
||||
"net/http"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// TestBackendTransportPoolsConnections guards the fix for the gateway->backend
|
||||
// connection churn. Every synchronous client call proxies to the single backend host,
|
||||
// so the REST client must widen the idle-connection pool past the default per-host cap
|
||||
// of 2 (http.DefaultMaxIdleConnsPerHost) — otherwise almost every request under load
|
||||
// opens a fresh TCP connection that then lingers in TIME_WAIT, burning gateway CPU and
|
||||
// exhausting ephemeral ports. Reverting to the default transport (`&http.Client{...}`
|
||||
// with no Transport) would silently reintroduce that, so assert the pool is widened.
|
||||
func TestBackendTransportPoolsConnections(t *testing.T) {
|
||||
c, err := New("http://backend.invalid", "localhost:9090", time.Second)
|
||||
if err != nil {
|
||||
t.Fatalf("New: %v", err)
|
||||
}
|
||||
defer func() { _ = c.Close() }()
|
||||
|
||||
tr, ok := c.http.Transport.(*http.Transport)
|
||||
if !ok {
|
||||
t.Fatalf("REST transport = %T, want a *http.Transport with a widened idle pool", c.http.Transport)
|
||||
}
|
||||
if tr.MaxIdleConnsPerHost <= http.DefaultMaxIdleConnsPerHost {
|
||||
t.Errorf("MaxIdleConnsPerHost = %d, want > default %d (else per-call connection churn)",
|
||||
tr.MaxIdleConnsPerHost, http.DefaultMaxIdleConnsPerHost)
|
||||
}
|
||||
}
|
||||
+12
-8
@@ -15,10 +15,12 @@ and prints a trip-report summary. It stays in the repo for repeats.
|
||||
2. **Drive** (edge protocol over h2c): assembles real 2–4 player games via the
|
||||
invitation flow (`invitation.create` → `invitation.accept`, no robots), then runs
|
||||
each player's turn loop — poll `game.state`, replay `game.history`, generate a legal
|
||||
**mid-ranked** move with the embedded `scrabble-solver`, and `game.submit_play`
|
||||
(or pass/exchange). A fraction of turns exercise nudge / chat / check-word / draft /
|
||||
profile-update / stats. Each player also holds a live `Subscribe` stream. The
|
||||
moderate ramp is **50 → 200 → 500** concurrent players, ~12 min per step.
|
||||
**mid-ranked** move with the embedded `scrabble-solver`, **compose it tile by tile with
|
||||
the debounced `game.evaluate` preview a real client fires** (the hottest gameplay call),
|
||||
persist a `draft.save`, and `game.submit_play` (or pass/exchange). A fraction of turns
|
||||
exercise nudge / chat / check-word / draft / profile-update / stats. Each player also
|
||||
holds a live `Subscribe` stream. The moderate ramp is **50 → 200 → 500** concurrent
|
||||
players, ~12 min per step. `--eval=false` drops the evaluate model for an A/B baseline.
|
||||
3. **Hammer**: drives `games.list` from one account far above the per-user rate limit
|
||||
to verify the limiter holds (`rate_limited` results) and measure its cost.
|
||||
4. **Report**: per-operation latency percentiles, throughput, result-code breakdown,
|
||||
@@ -72,6 +74,8 @@ Key `run` flags (env in parentheses):
|
||||
| `--games-per-player` | `0` (random 3–5) | target concurrent games per player |
|
||||
| `--tick` | `800ms` | per-player op cadence (keeps a player under the per-user limit) |
|
||||
| `--secondary-prob` | `0.08` | chance per tick of a non-move op |
|
||||
| `--eval` | `true` | model the per-tile `game.evaluate` preview (the gameplay hot path); `false` reproduces the pre-evaluate harness |
|
||||
| `--eval-recon` | `1` | extra full-composition evaluate re-previews per play (reconsideration), beyond one per placed tile |
|
||||
| `--hammer-workers` / `--hammer-dur` | `20` / `15s` | gateway-hammer (0 workers disables) |
|
||||
| `--reset` / `--cleanup` | `false` | delete harness rows before / after the run |
|
||||
|
||||
@@ -93,11 +97,11 @@ runs unconditionally. Use an **absolute** path (here via `$PWD`): `go test ./loa
|
||||
runs each package from its own directory, so a relative `BACKEND_DICT_DIR` would not
|
||||
resolve.
|
||||
|
||||
## Trip reports
|
||||
## Trip report
|
||||
|
||||
The two stress passes are written up in the repo: the early pass in
|
||||
[`REPORT-R2.md`](REPORT-R2.md) and the final, tuned pass in
|
||||
[`REPORT-R7.md`](REPORT-R7.md).
|
||||
The stress findings — the final run, the `game.evaluate` hot-path model, the
|
||||
gateway→backend connection-pool fix, and the revised sizing — are written up in
|
||||
[`REPORT.md`](REPORT.md).
|
||||
|
||||
## Caveat
|
||||
|
||||
|
||||
@@ -1,162 +0,0 @@
|
||||
# R2 — early stress-run trip report
|
||||
|
||||
The early stress pass for `PRERELEASE.md` R2. It exercises the system through the
|
||||
**edge protocol** with the `scrabble/loadtest` harness, to surface logic/concurrency
|
||||
bugs and capture a resource baseline that feeds R3 (edge hardening), R6 (refactor) and
|
||||
R7 (final tuning). Pass bar: **diagnostic** — the run "passes" by completing without the
|
||||
harness crashing; findings are recorded below, not gated.
|
||||
|
||||
## Method
|
||||
|
||||
- **Driver:** the `scrabble/loadtest` module, run as a one-shot container on the
|
||||
`scrabble-internal` docker network (reaching `postgres:5432` and `gateway:8081`
|
||||
directly, bypassing the host→gateway hairpin).
|
||||
- **Seed:** 10 000 durable + 1 000 guest accounts with pre-created sessions written
|
||||
directly to Postgres (token hash matches `backend/internal/session`), so the driver
|
||||
authenticates without the per-IP-limited auth ops.
|
||||
- **Games:** assembled through the real **invitation** flow (`invitation.create` →
|
||||
`invitation.accept`), 2–4 players each, no robots; variants spread over
|
||||
scrabble_en / scrabble_ru / erudit_ru.
|
||||
- **Play:** each virtual player holds a live `Subscribe` stream and, per tick, polls
|
||||
`game.state`, replays `game.history` and submits a **mid-ranked** legal move generated
|
||||
locally by the embedded `scrabble-solver` (the edge carries no board), or
|
||||
passes/exchanges; a fraction exercise nudge / chat / check-word / draft / profile /
|
||||
stats. A separate **gateway-hammer** floods `games.list` from one account.
|
||||
- **Scale:** moderate ramp **50 → 200 → 500** concurrent players, 10 min/step (the
|
||||
agreed moderate profile; harness and contour share this host's CPU).
|
||||
- **Resource capture:** `docker stats` (docker API) sampled every 28 s for per-container
|
||||
CPU/memory; Prometheus for edge latency/throughput, `postgres_exporter` internals and
|
||||
per-service Go runtime metrics.
|
||||
|
||||
## Run configuration
|
||||
|
||||
```
|
||||
loadtest run --durable 10000 --guest 1000 --steps 50,200,500 --step-dur 10m \
|
||||
--tick 800ms --hammer-workers 20 --hammer-dur 15s --cleanup
|
||||
```
|
||||
|
||||
Date: 2026-06-09. Contour: the R1-baseline schema, freshly deployed with the R2
|
||||
exporters. Seeded population removed by `--cleanup` afterwards.
|
||||
|
||||
## Findings
|
||||
|
||||
### Validated (fixed within R2)
|
||||
- **Harness draft payload.** `draft.save` first returned `bad_request`: the backend
|
||||
draft DTO's `rack_order` is a string (the harness sent `[]`). Fixed → `ok`.
|
||||
- **Harness profile marker.** `profile.update` first returned `invalid_profile`: the
|
||||
editable-display-name validator (`backend/internal/account/profile.go`) forbids digits
|
||||
and colons, but the seed marker was `lt:…`. Switched the marker to a distinctive
|
||||
letters-only string → `ok`. Cleanup still matches it.
|
||||
|
||||
### By-design behaviour (correctly exercised, not bugs)
|
||||
- **`chat_not_your_turn`** — chat is gated to the sender's turn
|
||||
(`backend/internal/social/chat.go`); off-turn posts are correctly rejected.
|
||||
- **`nudge_own_turn`** — you nudge the player whose turn it is, so a nudge on your own
|
||||
turn is correctly rejected. The harness nudges/chats at random ticks, so a share of
|
||||
these codes is expected.
|
||||
|
||||
### Observability gap (key R7 input)
|
||||
- **cAdvisor yields only the root cgroup on the contour host.** Its docker factory
|
||||
registers, but per-container init fails — `failed to identify the read-write layer ID
|
||||
… /rootfs/var/lib/docker/image/overlayfs/…: no such file or directory` — because this
|
||||
host's `/var/lib/docker` is a **separate XFS mount** not visible under cAdvisor's
|
||||
`/rootfs` bind (the existing galaxy deployment on the same host has the same
|
||||
limitation). So the **Scrabble — Resources** dashboard's per-container panels are empty
|
||||
here, and per-container CPU/RSS for this run was captured via `docker stats` instead.
|
||||
Postgres internals (`postgres_exporter`) and per-service Go runtime metrics
|
||||
(`go_*` by `service_name`) work. **Recommendation for R7:** adopt the otelcol
|
||||
**`docker_stats`** receiver (already the contrib image) — it reads per-container stats
|
||||
via the docker API with no cgroup dependency — and/or run the final pass on hardware
|
||||
where cAdvisor resolves containers. (Decision to confirm with the owner.)
|
||||
|
||||
### Run results
|
||||
|
||||
The ramp ran clean to 500 players with no harness crash, no deadlock and
|
||||
`stream errors: 0`; cleanup removed all 11 000 seeded accounts (and their ~941 games).
|
||||
|
||||
- **Ramp:** step 1 = 50 players / 90 games, step 2 = 200 / 282, step 3 = 500 / 569.
|
||||
- **Volume (30 min):** 1.20 M total edge calls, 659 req/s average. Real gameplay at
|
||||
scale: **48 870 committed plays**, 52 772 `your_turn` + 159 631 `opponent_moved`
|
||||
events, **2 798 games finished**.
|
||||
- **Latency under load (peak, step 3):** `game.state` p50 ≈ 100 ms, p90/p99 in the
|
||||
200–500 ms buckets, max 849 ms; `game.submit_play` similar (p99 ≤ 500 ms, max 490 ms).
|
||||
Lobby ops stayed fast (invitation/games.list p99 ≤ 10 ms).
|
||||
- **Rate limiter holds.** The gateway-hammer sent 522 667 `games.list` from one account;
|
||||
**522 486 (99.97 %) were `rate_limited`**, only 135 `ok` (the burst). Rejections are
|
||||
cheap — p99 = 2 ms — and the gateway sustained ~16 k req/s of rejections during the
|
||||
flood. The per-user limiter behaves as designed (R3 input: the cost is negligible).
|
||||
|
||||
**Top finding — `transport_error` under saturation.** At 500 players ~14 % of
|
||||
`game.state` calls (72 429 / 519 067) and a few % of the other ops returned a Connect
|
||||
`transport_error` (not a domain code). It correlates with the CPU saturation below: the
|
||||
backend/gateway are pinned near one core each while the host also runs the 86 %-core
|
||||
harness, so the edge sheds load (resets/timeouts) at the knee. It is **amplified by a
|
||||
harness artifact** — all 500 virtual players multiplex over a *single* shared
|
||||
`http2.Transport`, so 500 persistent `Subscribe` streams plus Execute calls press on one
|
||||
HTTP/2 connection's concurrent-stream limit; real clients each use their own connection.
|
||||
**Actions:** R7 harness — give each player (or a pool) its own transport, and run on
|
||||
hardware not shared with the contour; R3 — confirm the gateway's h2c
|
||||
`MaxConcurrentStreams` and edge timeouts are sized for many persistent streams.
|
||||
|
||||
**Minor findings:**
|
||||
- `unauthenticated` on a tiny share (188 / 519 067 `game.state`, ~0.04 %) — transient
|
||||
session-resolve failures under load; worth a glance in R3 but not material.
|
||||
- one `internal` on `game.pass` (1 / 4 788).
|
||||
- `game_finished` dominates `chat.nudge`/`chat.post` (≈ 3 900 each): the harness keeps
|
||||
secondary ops on games that already ended. Harness refinement — drop finished games
|
||||
from the rotation (R7).
|
||||
- `nudge_own_turn` / `chat_not_your_turn` / `nudge_too_soon` are the expected turn/rate
|
||||
gates, correctly exercised.
|
||||
|
||||
## Resource baseline
|
||||
|
||||
Per-container peak during step 3 (500 players), from `docker stats`:
|
||||
|
||||
| container | peak CPU | memory |
|
||||
|-----------|---------:|-------:|
|
||||
| scrabble-backend | **99 %** (~1 core) | 91 MiB |
|
||||
| scrabble-gateway | **93 %** | 76 MiB |
|
||||
| scrabble-postgres | **90 %** | 69 MiB |
|
||||
| scrabble-loadtest (harness) | **86 %** | 42 MiB |
|
||||
| scrabble-otelcol | 10 % | 110 MiB |
|
||||
| scrabble-tempo | 9 % | 446 MiB |
|
||||
| prometheus / postgres-exporter | ~0 % | 46 / 16 MiB |
|
||||
|
||||
- **The contour is CPU-bound at 500 concurrent players:** backend, gateway and Postgres
|
||||
each saturate ~1 core (single-instance MVP config), so the system draws ~3 cores at
|
||||
this scale; memory is modest (≤ 100 MiB per Go service). This is the sizing input for
|
||||
R7 (pool sizes, GOMAXPROCS, container limits) and the prod cutover.
|
||||
- **Caveat:** the harness itself peaked at **86 % of a core** on the *same host*, so the
|
||||
step-3 latency and `transport_error` figures are pessimistic — the contour competed
|
||||
with the generator for CPU. A clean ceiling needs separate hardware (R7).
|
||||
- **Postgres:** peak 28 backend connections, ~5 581 commits/s at the peak, **100 % cache
|
||||
hit ratio** (no disk reads) — the DB was comfortable; CPU, not I/O, is its limit here.
|
||||
- **Goroutines:** backend 638, gateway **1 698** (it holds the 500 `Subscribe` streams +
|
||||
per-request goroutines), telegram 49 — all stable, no leak across the ramp.
|
||||
|
||||
## Recommendations feeding later phases
|
||||
- **R3 (edge hardening):** the per-user limiter holds (99.97 % rejected, p99 2 ms) — add
|
||||
the per-IP body-size cap on top. Investigate the **~14 % `transport_error` on
|
||||
`game.state` at 500 players**: confirm the gateway h2c `MaxConcurrentStreams` and edge
|
||||
read/write timeouts are sized for many persistent `Subscribe` streams, and glance at the
|
||||
~0.04 % transient `unauthenticated` resolves under load.
|
||||
- **R6 (refactor):** no logic bug forced a code change beyond the two harness-payload
|
||||
fixes; the run surfaced no deadlock or goroutine leak across the ramp.
|
||||
- **R7 (final tuning + stress):** (1) fix the per-container observability gap — adopt the
|
||||
otelcol `docker_stats` receiver so Grafana shows per-container CPU/RSS on the contour;
|
||||
(2) refine the harness — per-player/pooled transports and dropping finished games from
|
||||
the rotation — and run on hardware **not** shared with the contour; (3) size pools /
|
||||
GOMAXPROCS / container limits from the CPU-bound peak (~1 core each for backend, gateway,
|
||||
Postgres at 500 players).
|
||||
|
||||
## Re-running
|
||||
|
||||
See [`README.md`](README.md). Briefly, from the repo root:
|
||||
|
||||
```sh
|
||||
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
||||
docker run --rm --name scrabble-loadtest --network scrabble-internal \
|
||||
-e POSTGRES_PASSWORD=… scrabble-loadtest run # add --reset on a re-run
|
||||
```
|
||||
|
||||
The harness stays in the repo for the R7 repeat.
|
||||
@@ -1,212 +0,0 @@
|
||||
# R7 — final stress-run trip report
|
||||
|
||||
The final pre-release stress pass for [`PRERELEASE.md`](../PRERELEASE.md) R7. It re-runs
|
||||
the R2 harness (`scrabble/loadtest`) against the **final, refactored system** on a
|
||||
freshly redeployed contour, to confirm the system holds at scale and to settle the
|
||||
resource sizing (container limits, `GOMAXPROCS`, pools, rate limits, log levels) before
|
||||
the Stage 18 prod cutover. Pass bar: **diagnostic + a tuning decision** — the run
|
||||
"passes" by completing cleanly; the per-container resource profile drives the tuning
|
||||
recorded below. Companion to the early pass, [`REPORT-R2.md`](REPORT-R2.md).
|
||||
|
||||
## What changed since the R2 pass
|
||||
|
||||
- **Harness — per-player transports.** Each virtual player now owns its `edge.Client`
|
||||
(its own `http2.Transport` / h2c connection carrying both its `Subscribe` stream and
|
||||
its `Execute` calls), instead of all players multiplexing over one shared transport.
|
||||
R2 traced the ~14 % `transport_error` on `game.state` at 500 players to that single
|
||||
shared connection's stream limit; per-player connections mirror real clients and
|
||||
remove the artifact, so this pass measures the system, not the harness.
|
||||
- **Harness — drop finished games.** `playTurn` reports a finished game and the player
|
||||
drops it from its rotation, so secondary ops stop hitting `game_finished` on ended
|
||||
games (the other R2 harness finding).
|
||||
- **Observability — otelcol `docker_stats`.** cAdvisor (which resolves only the root
|
||||
cgroup on this host — separate-XFS `/var/lib/docker`) is replaced by the otelcol
|
||||
`docker_stats` receiver, reading per-container CPU/memory/network from the Docker API.
|
||||
Per-container panels now populate on the contour host. (`api_version` pinned to 1.44;
|
||||
the daemon's minimum is 1.40.)
|
||||
- **Contour — container limits + `GOMAXPROCS`.** `deploy.resources.limits` now bound
|
||||
every service; the Go services pin `GOMAXPROCS` to their CPU limit so the runtime
|
||||
matches the cgroup quota. Starting values were generous over the R2 peak; this pass
|
||||
validates them and settles the agreed sizing (below).
|
||||
|
||||
## Method
|
||||
|
||||
Unchanged from R2 except for the per-player transports and the dropped-finished-games
|
||||
refinement above:
|
||||
|
||||
- **Driver:** the `scrabble/loadtest` module, run as a one-shot container on the
|
||||
`scrabble-internal` docker network (reaching `postgres:5432` / `gateway:8081`
|
||||
directly), capped at `--cpus 3` so the contour keeps the host's spare cores.
|
||||
- **Seed:** 10 000 durable + 1 000 guest accounts with pre-created sessions written
|
||||
straight to Postgres (token hash matches `backend/internal/session`).
|
||||
- **Games:** assembled through the real **invitation** flow, 2–4 players each, no
|
||||
robots; variants over scrabble_en / scrabble_ru / erudit_ru.
|
||||
- **Play:** each player holds a live `Subscribe` stream and, per tick, polls
|
||||
`game.state`, replays `game.history` and submits a **mid-ranked** legal move generated
|
||||
locally by the embedded `scrabble-solver`, or passes / exchanges; a fraction exercise
|
||||
nudge / chat / check-word / draft / profile / stats. A separate **gateway-hammer**
|
||||
floods `games.list` from one account.
|
||||
- **Scale:** the same moderate ramp **50 → 200 → 500** concurrent players, 10 min/step.
|
||||
- **Resource capture:** `docker stats` (docker API) sampled every ~20 s for per-container
|
||||
CPU/memory; the otelcol **`docker_stats`** receiver → Prometheus → the Grafana
|
||||
**Scrabble — Resources** dashboard for the same per-container series; `postgres_exporter`
|
||||
internals and per-service Go runtime metrics.
|
||||
|
||||
## Run configuration
|
||||
|
||||
```
|
||||
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
||||
-e POSTGRES_PASSWORD=… scrabble-loadtest \
|
||||
run --durable 10000 --guest 1000 --steps 50,200,500 --step-dur 10m \
|
||||
--tick 800ms --hammer-workers 20 --hammer-dur 15s --reset --cleanup
|
||||
```
|
||||
|
||||
Date: 2026-06-10. Contour: the R1-baseline schema, freshly redeployed with the R7
|
||||
container limits / `GOMAXPROCS` (backend/gateway/postgres capped at 2 cores + 512 MiB,
|
||||
`GOMAXPROCS=2`) and the `docker_stats` observability. Seeded population removed by
|
||||
`--cleanup` afterwards.
|
||||
|
||||
## Findings
|
||||
|
||||
The ramp ran clean to 500 players — no harness crash, no deadlock, `stream errors: 0` —
|
||||
and cleanup removed all 11 000 seeded accounts.
|
||||
|
||||
- **Volume (1827 s):** 821 680 edge calls (449.7 req/s incl. the hammer). Real gameplay
|
||||
at scale: **50 916 committed plays**, 4 817 passes, 2 931 games finished; 165 755
|
||||
`opponent_moved` + 54 864 `your_turn` events.
|
||||
- **The per-player transport fix worked.** `game.state` returned `transport_error` on
|
||||
**3 173 / 127 403 = 2.49 %** of calls — down from R2's ~14 % on the same step. Other
|
||||
ops were lower still (`game.history` 0.43 %, `game.submit_play` 0.28 %). The residual
|
||||
is the gateway bursting into its 2-core cap (see the profile below), not the harness.
|
||||
- **Dropping finished games worked.** `game_finished` on `chat.nudge` / `chat.post` fell
|
||||
to **35 / 36** (R2: ≈ 3 900 each) — secondary ops no longer hammer ended games.
|
||||
- **The limiter holds.** The gateway-hammer sent 565 152 `games.list`; **564 979
|
||||
(99.97 %) were `rate_limited`** (154 ok burst, 19 deadline), p99 = 2 ms, ~309 req/s of
|
||||
rejections sustained — unchanged from R2.
|
||||
- **Latency (peak):** `game.state` p50 ≈ 100 ms, p99 in the 2000 ms bucket (max 2549 ms);
|
||||
`game.submit_play` p50 100 / p99 1000 ms bucket. Lobby ops stayed fast
|
||||
(invitation / games.list p99 ≤ 10 ms). The p99 tail correlates with the gateway
|
||||
burst-throttling, not the backend (which stayed at ~0.85 core).
|
||||
|
||||
## Resource profile
|
||||
|
||||
Per-container peak during step 3 (500 players), with the R7 starting limits in force
|
||||
(backend/gateway/postgres capped at 2 cores / 512 MiB). Two CPU columns: `docker stats`
|
||||
samples a ~1 s window (catches bursts); the otelcol `docker_stats` receiver averages over
|
||||
its 30 s collection interval (smooths them) — they agree within sampling error, which
|
||||
validates the new observability path.
|
||||
|
||||
| container | CPU burst (1 s) | CPU sustained (30 s) | CPU cap | mem peak | mem cap |
|
||||
|-----------|----------------:|---------------------:|--------:|---------:|--------:|
|
||||
| scrabble-gateway | **217 %** (at cap) | ~145 % | 200 % | 167 MiB | 512 MiB |
|
||||
| scrabble-postgres | 138 % | ~153 % | 200 % | 117 MiB | 512 MiB |
|
||||
| scrabble-backend | 85 % | ~89 % | 200 % | 116 MiB | 512 MiB |
|
||||
| scrabble-tempo | 33 % | — | (none) | **1024 MiB** (at cap) | 1024 MiB |
|
||||
| scrabble-otelcol | 11 % | — | (none) | 131 MiB | 512 MiB |
|
||||
| scrabble-loadtest (harness) | 157 % | — | 300 % | 369 MiB | — |
|
||||
|
||||
- **The gateway is the binding constraint.** With one h2c connection per player it draws
|
||||
~1.45 cores sustained and **bursts to its 2-core cap** at 500 players, throttling
|
||||
briefly — the source of the 2.49 % `transport_error`. R2 saw only ~0.93 core because
|
||||
all 500 players shared one connection; the +~0.5 core is the realistic per-connection
|
||||
overhead (500 separate HTTP/2 connections). This is a sizing fact, not a regression.
|
||||
- **backend is over-provisioned** (~0.85 core vs a 2-core cap); **postgres** (~1.4 cores)
|
||||
has headroom; both stayed ≤ 120 MiB.
|
||||
- **tempo reached its 1 GiB memory cap** (R2: 446 MiB) — an OOM risk under sustained
|
||||
tracing.
|
||||
- **Postgres backends peaked at 28**, with the backend pool at its `MaxOpenConns=25` cap.
|
||||
Cache hit stayed ~100 % (no disk reads); CPU, not I/O, is the limit.
|
||||
- **docker log volume (30 min):** backend 14.2 MiB, gateway 4.6 MiB, postgres 0.04 MiB —
|
||||
the backend's per-request latency line at info dominates, and json-file logs had no
|
||||
rotation.
|
||||
|
||||
## Tuning applied
|
||||
|
||||
Agreed from the profile (all in `deploy/docker-compose.yml`; no code change — the pool
|
||||
is already env-driven):
|
||||
|
||||
| knob | from | to | why |
|
||||
|------|------|----|-----|
|
||||
| gateway CPU + `GOMAXPROCS` | 2 cores / 2 | **3 cores / 3** | it bursts into the 2-core cap at 500 players (the 2.49 % `transport_error`); 3 absorbs the bursts |
|
||||
| tempo memory | 1 GiB | **2 GiB** | it reached the 1 GiB cap (OOM risk) |
|
||||
| backend `MAX_OPEN_CONNS` | 25 | **40** | the pool sat at its 25-conn cap at peak; headroom trims the p99 tail |
|
||||
| docker logs | unbounded | **json-file 10m × 3** | bound the ~14 MiB / 30 min backend log; level stays `info` |
|
||||
|
||||
Left as-is: backend / postgres at 2 cores / 512 MiB (peak ~0.85 / ~1.4 cores — headroom
|
||||
is cheap on the shared host); the per-user rate limiter and `h2cMaxConcurrentStreams=250`
|
||||
(per-connection now, ~1 stream each — ample) and cache TTLs (no pressure observed).
|
||||
|
||||
### Validation re-run
|
||||
|
||||
Re-running the **same gradual ramp** (50 → 200 → 500) on the tuned contour confirms the
|
||||
fix:
|
||||
|
||||
- **`game.state` `transport_error` fell to 0.72 %** (853 / 119 051), down from 2.49 % at
|
||||
2 cores. The latency tail also improved — p99 in the 1000 ms bucket, max 1220 ms (was
|
||||
the 2000 ms bucket, max 2549 ms).
|
||||
- The **gateway peaked at ~2 cores** (≈196 % on the 30 s gauge) — now comfortably **under
|
||||
the 3-core cap**, so it no longer throttles. backend ~1 core, postgres ~1.3 cores.
|
||||
- **tempo peaked at ~1.27 GiB** — under the new 2 GiB cap (it would have OOM-ed at 1 GiB).
|
||||
- Drop-finished still holds (`game_finished` on chat 41/42); the limiter still rejects
|
||||
99.97 % of the hammer at p99 2 ms; `stream errors: 0`.
|
||||
|
||||
A separate **burst stress** (a single 100 → 500 jump — 400 players connecting at once)
|
||||
**pegged the gateway at 3 cores** (≈296 % sustained) and pushed `game.state`
|
||||
`transport_error` to 9.27 %. The gateway is **connection-CPU-bound and bursty**: average
|
||||
load is ~1 core, but a mass-simultaneous connection storm saturates whatever single-node
|
||||
cap it is given. Real arrivals are gradual (the canonical run), where 3 cores has
|
||||
headroom; the lever for a true arrival spike is **horizontal scaling**, not more cores per
|
||||
node — carried into the prod recommendation below.
|
||||
|
||||
## Prod-sizing recommendation (Stage 18)
|
||||
|
||||
The contour is **CPU-bound and gateway-led** at 500 concurrent players. Carry these to the
|
||||
prod contour env (the same compose, `PROD_*` values):
|
||||
|
||||
- **gateway: ≥ 3 cores** per ~500 concurrent players, `GOMAXPROCS` pinned to the limit —
|
||||
it scales with the **connection count**, not just the request rate; beyond one node's
|
||||
worth, scale the gateway **horizontally** rather than vertically.
|
||||
- **backend: ~1–2 cores**, pool 40 — comfortable; the work is light per request.
|
||||
- **postgres: ~2 cores / ≥ 512 MiB** — ~1.4 cores at 500 players, 100 % cache hit.
|
||||
- **tempo: ≥ 2 GiB**; the Go services run under ~170 MiB (256 MiB would suffice, 512 is
|
||||
safe); pin `GOMAXPROCS` to each CPU limit; keep json-file rotation.
|
||||
- Memory is not the constraint anywhere; CPU is.
|
||||
|
||||
### VPS / VDS sizing (single-host contour)
|
||||
|
||||
The whole contour (the app + the observability stack) runs on one host via
|
||||
`docker-compose`. The tiers below are grounded in the R7 profile (**≈5.5 cores / ≈2.5 GiB
|
||||
RAM peak at 500 concurrent players**; ≈0.5 GiB idle) and the **measured** on-disk
|
||||
footprint: prod images ≈2.4 GB; the Tempo volume **3.1 GB at 72 h** retention; Prometheus
|
||||
≈1–2 GB at 15 d; the game DB 23 MiB and growing with history. CPU and disk grow; RAM has
|
||||
the most slack.
|
||||
|
||||
| tier | CPU | RAM | disk | handles |
|
||||
|------|-----|-----|------|---------|
|
||||
| **Minimum** | 2 cores | 2 GiB | 20 GiB | ~up to ~150 concurrent; lower the compose limits (gateway 1.5 / backend·postgres 1 / tempo 1 GiB) to fit the box |
|
||||
| **Average** (reasonable load) | 4 cores | 4 GiB | 40 GiB | ~300–400 concurrent comfortably; the tested 500 with occasional gateway burst-throttling |
|
||||
| **Maximum** (worry-free) | 8 cores | 8 GiB | 80 GiB | 500+ concurrent with full gateway burst headroom (its 3-core cap) + room to grow; the compose limits fit as-is |
|
||||
|
||||
- The per-service limits in `docker-compose.yml` are tuned for the **Average/Maximum**
|
||||
target (the gateway alone caps at 3 cores). On the **Minimum** tier, scale them down to
|
||||
match the host or the caps over-subscribe it.
|
||||
- **Disk is dominated by observability retention + DB growth.** Tempo (72 h traces) and
|
||||
Prometheus (15 d metrics) are the main levers — shorten the windows (or move Tempo to
|
||||
object storage) to cut disk; Postgres grows with game history, so budget for months of
|
||||
it; container logs are already capped (json-file 10m × 3 ≈ 30 MiB each).
|
||||
- **RAM** rarely binds: the contour peaks ≈2.5 GiB at 500 players and the sum of all
|
||||
configured limits is ≈5.6 GiB, so 8 GiB never strains.
|
||||
- Beyond one host's worth of players, scale the **gateway horizontally** (it is
|
||||
connection-CPU-bound) rather than ordering an ever-bigger box.
|
||||
|
||||
## Re-running
|
||||
|
||||
See [`README.md`](README.md). Briefly, from the repo root:
|
||||
|
||||
```sh
|
||||
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
||||
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
||||
-e POSTGRES_PASSWORD=… scrabble-loadtest run --reset --cleanup
|
||||
```
|
||||
|
||||
The harness stays in the repo for future repeats.
|
||||
@@ -0,0 +1,194 @@
|
||||
# loadtest — stress trip report
|
||||
|
||||
The pre-release stress write-up for [`PRERELEASE.md`](../PRERELEASE.md). It drives the
|
||||
`scrabble/loadtest` harness against a freshly redeployed test contour to confirm the
|
||||
system holds at scale and to settle resource sizing before the prod cutover. The harness
|
||||
stays in the repo for repeats; see [`README.md`](README.md) for how to run it.
|
||||
|
||||
This report supersedes the earlier per-phase notes. The harness has been through three
|
||||
passes: an early diagnostic, a tuning pass that sized container limits / `GOMAXPROCS`, and
|
||||
this final pass — which **added the per-tile `game.evaluate` preview to the model** (the
|
||||
hottest real gameplay call, previously unmodelled) and, with it, surfaced and fixed the
|
||||
**gateway→backend connection-pool bottleneck** described below. The numbers here are from
|
||||
that final pass.
|
||||
|
||||
## What it models
|
||||
|
||||
The harness seeds a large account population with pre-created sessions directly in
|
||||
Postgres, then drives virtual players through the **gateway edge protocol** (h2c) in real
|
||||
games assembled via the invitation flow. Each player owns its own `edge.Client` (its own
|
||||
h2c connection, like a real client), holds a live `Subscribe` stream, and per tick polls
|
||||
`game.state`, replays `game.history`, generates a legal **mid-ranked** move with the
|
||||
embedded `scrabble-solver`, and submits it (or passes/exchanges). A fraction of ticks
|
||||
exercise nudge / chat / check-word / draft / profile / stats. A separate **gateway-hammer**
|
||||
floods `games.list` to verify the rate limiter.
|
||||
|
||||
### The evaluate hot path (this pass)
|
||||
|
||||
A real client previews every tentative play as the user arranges tiles: the UI fires a
|
||||
debounced `game.evaluate` (legality + score) on each placement change while it is the
|
||||
player's turn. Over a single composed word that is **several evaluate calls per turn** —
|
||||
far more than the one `submit_play` — so `game.evaluate` is the single hottest gameplay
|
||||
request at scale. The earlier passes did not model it at all (they submitted directly),
|
||||
which understated the real load.
|
||||
|
||||
This pass models it: when a player composes a play of *K* newly-placed tiles, it fires one
|
||||
`evaluate` per landed tile (a growing prefix of the tiles), plus a small number of
|
||||
full-composition re-previews for reconsideration, spaced by a human-paced gap (the client's
|
||||
250 ms debounce), then one `draft.save`, then `submit_play`. `--eval=false` reproduces the
|
||||
pre-evaluate harness for an A/B baseline; `--eval-recon` tunes the reconsideration count.
|
||||
|
||||
`game.check_word` is a *different*, manual "look this word up" panel (throttled, on demand)
|
||||
— not the per-tile call — and is exercised separately as a secondary op.
|
||||
|
||||
## Final run (eval-on, after the connection-pool fix)
|
||||
|
||||
Contour: backend / postgres capped at 2 cores / 512 MiB (`GOMAXPROCS=2`), gateway at
|
||||
3 cores / 512 MiB (`GOMAXPROCS=3`), per the tuned `deploy/docker-compose.yml`. Gradual ramp
|
||||
**50 → 200 → 500** concurrent players, 4 min/step, `--tick 800ms`, gateway-hammer on. The
|
||||
harness ran as a one-shot container on `scrabble-internal`, capped at `--cpus 3`. The DB was
|
||||
wiped before the run (`DROP SCHEMA backend CASCADE`); the seeded population was removed by
|
||||
`--cleanup` afterwards.
|
||||
|
||||
Per-operation results at the 500-player peak (740 s, gameplay rows; the hammer row is the
|
||||
limiter probe):
|
||||
|
||||
| operation | count | req/s | p50 | p99 | max | notes |
|
||||
|-----------|------:|------:|----:|----:|----:|-------|
|
||||
| game.evaluate | 85 721 | 115.9 | 1 ms | 200 ms | 193 ms | **the hot path** — all ok |
|
||||
| game.state | 115 926 | 156.7 | 100 ms | 200 ms | 260 ms | transport_error 86 (0.07 %) |
|
||||
| game.history | 22 258 | 30.1 | 5 ms | 100 ms | 195 ms | all ok |
|
||||
| draft.save | 23 031 | 31.1 | 2 ms | 200 ms | 194 ms | all ok |
|
||||
| game.submit_play | 21 704 | 29.3 | 1 ms | 200 ms | 274 ms | ok 3 902; not_your_turn / illegal_play are concurrent-play races (see caveat) |
|
||||
| hammer:games.list | 522 756 | 706.7 | 1 ms | 2 ms | 53 ms | **99.97 % rate_limited** — limiter holds |
|
||||
|
||||
- **Volume:** 802 200 total edge calls (1 084 req/s incl. the hammer; ~377 req/s of real
|
||||
gameplay). `stream errors: 0`. Live events: 11 199 `opponent_moved`, 4 153 `your_turn`.
|
||||
- **`game.evaluate` is the dominant gameplay write-path call** at ~116 req/s — second only
|
||||
to the `game.state` poll — and it is cheap: p50 1 ms, effectively zero errors. The backend
|
||||
serves it straight from the in-memory live-game cache; on a warm hit it skips the database
|
||||
entirely (see *Postgres read path* below, which halved its p99 to 100 ms).
|
||||
- **Latency stayed healthy** under the heavier evaluate load: every gameplay op p99 ≤ 200 ms.
|
||||
- **The limiter holds** unchanged: 99.97 % of the hammer rejected at p99 2 ms.
|
||||
|
||||
### Peak CPU (500 players)
|
||||
|
||||
| container | CPU peak | cap |
|
||||
|-----------|---------:|----:|
|
||||
| scrabble-postgres | **165 %** (~1.65 cores) | 200 % |
|
||||
| scrabble-backend | 77 % (~0.77 core) | 200 % |
|
||||
| scrabble-gateway | **26 %** (~0.26 core) | 300 % |
|
||||
| scrabble-loadtest (harness) | 42 % | 300 % |
|
||||
|
||||
Memory stayed modest everywhere (Go services ≤ ~90 MiB). **Postgres is now the busiest
|
||||
service** — it has headroom (1.65 of 2 cores) but is the scaling axis. The gateway, after
|
||||
the fix below, is near-idle.
|
||||
|
||||
## The headline finding: gateway→backend connection churn
|
||||
|
||||
The gateway proxies every synchronous client call to the single backend host over REST.
|
||||
Its backend HTTP client used the default transport, whose **`MaxIdleConnsPerHost` is 2**
|
||||
(`http.DefaultMaxIdleConnsPerHost`). So the gateway kept only **2** keep-alive connections
|
||||
to the backend and opened — then closed — a fresh TCP connection for almost every other
|
||||
call. Measured at the gateway's network namespace:
|
||||
|
||||
| | gateway→backend sockets |
|
||||
|---|---|
|
||||
| before (eval-on, 500 players) | **TIME_WAIT ≈ 26 500**, ESTABLISHED 2 |
|
||||
| after (eval-on, 500 players) | TIME_WAIT ≈ 0 (steady state), **ESTABLISHED ≈ 225 (reused)** |
|
||||
|
||||
26 500 TIME_WAIT sockets is the connection **churn**: ~440 new connections per second,
|
||||
each a full TCP handshake + teardown, the socket then lingering 60 s. That count sits right
|
||||
under the ~28 000 ephemeral-port ceiling — the latent cliff that produced the residual
|
||||
`transport_error` the earlier passes chased on the *client* side (h2c streams) but never
|
||||
eliminated, because the real cause was here, on the *backend* side.
|
||||
|
||||
The fix is one custom `http.Transport` with a wide idle pool
|
||||
(`gateway/internal/backendclient/client.go`, `backendMaxIdleConns`). Before / after, same
|
||||
eval-on workload at 500 players:
|
||||
|
||||
| metric | before fix | after fix |
|
||||
|--------|-----------:|----------:|
|
||||
| gateway→backend TIME_WAIT | ~26 500 | **~0** |
|
||||
| gateway CPU peak | **175 %** (~1.75 cores) | **26 %** (~0.26 core) |
|
||||
| game.state p99 | 500 ms | 200 ms |
|
||||
|
||||
**The churn was burning ~1.5 gateway cores of pure connection setup/teardown.** Removing it
|
||||
cut peak gateway CPU ~7× and erased the port-exhaustion cliff. The backend and postgres CPU
|
||||
are unchanged — they do the real work; only the gateway's wasted overhead disappeared. The
|
||||
pool settles at ~225 live connections at 500 players; the constant is set to 512 for ~2×
|
||||
headroom.
|
||||
|
||||
## Sizing — why the old "≈150 concurrent / 2-core" figure was a bug, not a floor
|
||||
|
||||
The earlier tuning pass concluded the gateway was the binding constraint — "size it for
|
||||
≥ 3 cores per 500 players, scale it horizontally" — and the single-host "minimum" tier
|
||||
topped out near ~150 concurrent. **That was sizing around the connection-churn bug.** The
|
||||
gateway drew ~1.75–3 cores not from proxying work but from churning backend connections;
|
||||
the backend behind it sat near-idle the whole time.
|
||||
|
||||
With the churn fixed, at **500 concurrent players** the app draws roughly:
|
||||
|
||||
- **gateway ≈ 0.26 core** (was ~3) — no longer the constraint,
|
||||
- **backend ≈ 0.77 core**,
|
||||
- **postgres ≈ 1.65 cores** — now the busiest, with headroom,
|
||||
|
||||
≈ **2.7 app cores total** (down from the ~5.5-core contour peak the tuning pass recorded,
|
||||
*and* under a heavier, more realistic workload that now includes `game.evaluate`). Postgres,
|
||||
not the gateway, is the scaling axis.
|
||||
|
||||
Revised single-host guidance (app + co-resident observability stack on one box):
|
||||
|
||||
| tier | CPU | RAM | handles |
|
||||
|------|-----|-----|---------|
|
||||
| **Minimum** | 2 cores | 2 GiB | comfortably the low hundreds of concurrent — the gateway no longer eats cores; postgres + the observability stack set the limit |
|
||||
| **Average** | 4 cores | 4 GiB | 500 concurrent with headroom |
|
||||
| **Maximum** | 8 cores | 8 GiB | 500+ with full burst headroom and room to grow |
|
||||
|
||||
The gateway's compose limit can drop well below its old 3 cores; it is now connection-pool
|
||||
bound, not connection-CPU bound. Memory was never the constraint. Disk is still dominated
|
||||
by observability retention (Tempo, Prometheus) + DB growth — unchanged from before.
|
||||
|
||||
## Postgres read path (warm-cache optimization)
|
||||
|
||||
Following this pass, `game.evaluate` no longer reads the database on the hot path. An
|
||||
active game is already resident in the in-memory live-game cache (mutated in place across
|
||||
moves, evicted only on finish), so the preview answers its seat-membership check from the
|
||||
cached immutable seat list and scores against the cached engine game — **no `GetGame` on a
|
||||
warm hit**. `GetGame` itself was also folded from two round-trips (game, then seats) into a
|
||||
single `LEFT JOIN`. Measured at 500 players, **`game.evaluate` p99 halved (200 → 100 ms)**
|
||||
and the per-operation query count dropped.
|
||||
|
||||
It did **not** cut postgres CPU, and the measurement says why: postgres is **write-bound**,
|
||||
not read-bound. `pg_stat_user_tables` puts the cost in the per-move `CommitMove`
|
||||
transaction (a `game_moves` insert plus `games` / `game_players` updates), the debounced
|
||||
`game_drafts` upserts (~60 k in one run), and the journal replays — not the cheap, indexed,
|
||||
fully-cached `GetGame` lookups this change removed (one re-run even committed 28 % more
|
||||
plays, whose extra writes masked the saved reads). Postgres also runs with headroom
|
||||
(~1.5 of 2 cores), and the gateway fix freed ~3 cores on the box, so the lever if postgres
|
||||
ever caps is **more cores** (it is CPU-bound, not I/O), not riskier write-path surgery. So
|
||||
this change is a latency / query-volume win, deliberately not a DB-CPU one.
|
||||
|
||||
## Caveat — harness fidelity
|
||||
|
||||
The harness's `not_your_turn` and `illegal_play` on `submit_play` are concurrent-play
|
||||
artifacts, not system errors: it generates a move from a locally replayed board, and a
|
||||
fast opponent (or a transport hiccup) can move between the state fetch and the submit,
|
||||
leaving the move out of turn or illegal on the now-changed board. A real client previews
|
||||
with `evaluate` and only submits a legal, in-turn play. These rejections are cheap domain
|
||||
outcomes (HTTP-ok with a stable code) and do not change the request *load*, which is what
|
||||
the run measures. The harness also shares the host CPU with the contour (capped with
|
||||
`--cpus`); a fully isolated ceiling on separate hardware remains future work.
|
||||
|
||||
## Re-running
|
||||
|
||||
From the repo root:
|
||||
|
||||
```sh
|
||||
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
||||
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
||||
-e POSTGRES_PASSWORD="$TEST_POSTGRES_PASSWORD" scrabble-loadtest run --reset --cleanup
|
||||
```
|
||||
|
||||
`--eval=false` reproduces the pre-evaluate baseline for comparison. The authoritative hard
|
||||
reset of the contour DB remains `DROP SCHEMA backend CASCADE` + a backend restart.
|
||||
@@ -73,6 +73,8 @@ func cmdRun(ctx context.Context, log *slog.Logger, args []string) error {
|
||||
gpp := fs.Int("games-per-player", 0, "target concurrent games per player (0 => random 3..5)")
|
||||
tick := fs.Duration("tick", 800*time.Millisecond, "per-player operation cadence")
|
||||
secProb := fs.Float64("secondary-prob", 0.08, "chance per tick of a non-move operation")
|
||||
eval := fs.Bool("eval", true, "model the per-tile evaluate preview (the realistic gameplay hot path); --eval=false reproduces the pre-evaluate harness for an A/B baseline")
|
||||
evalRecon := fs.Int("eval-recon", 1, "extra full-composition evaluate re-previews per play (reconsideration), beyond one per placed tile")
|
||||
hammerWorkers := fs.Int("hammer-workers", 20, "gateway-hammer concurrent callers (0 disables)")
|
||||
hammerDur := fs.Duration("hammer-dur", 15*time.Second, "gateway-hammer duration")
|
||||
reset := fs.Bool("reset", false, "delete prior harness rows before seeding")
|
||||
@@ -117,6 +119,7 @@ func cmdRun(ctx context.Context, log *slog.Logger, args []string) error {
|
||||
cfg := scenario.RealisticConfig{
|
||||
Steps: steps, StepDur: *stepDur, GamesPerPlayer: *gpp,
|
||||
Tick: *tick, SecondaryProb: *secProb,
|
||||
Eval: *eval, EvalRecon: *evalRecon,
|
||||
}
|
||||
if err := drv.RunRealistic(ctx, pool, cfg); err != nil && !errors.Is(err, context.Canceled) {
|
||||
return err
|
||||
|
||||
@@ -24,6 +24,7 @@ const (
|
||||
msgSubmitPlay = "game.submit_play"
|
||||
msgPass = "game.pass"
|
||||
msgExchange = "game.exchange"
|
||||
msgEvaluate = "game.evaluate"
|
||||
msgState = "game.state"
|
||||
msgHistory = "game.history"
|
||||
msgGamesList = "games.list"
|
||||
|
||||
@@ -63,6 +63,33 @@ func submitPlay(gameID string, tiles []PlayTile) []byte {
|
||||
return b.FinishedBytes()
|
||||
}
|
||||
|
||||
// evalReq builds an EvalRequest payload (game id plus the tentative newly-placed tiles).
|
||||
// It mirrors submitPlay's shape — the backend infers the play's orientation the same way —
|
||||
// so a preview previews exactly what submitting those tiles would score.
|
||||
func evalReq(gameID string, tiles []PlayTile) []byte {
|
||||
b := flatbuffers.NewBuilder(256)
|
||||
gid := b.CreateString(gameID)
|
||||
offs := make([]flatbuffers.UOffsetT, len(tiles))
|
||||
for i, t := range tiles {
|
||||
fb.PlayTileStart(b)
|
||||
fb.PlayTileAddRow(b, int32(t.Row))
|
||||
fb.PlayTileAddCol(b, int32(t.Col))
|
||||
fb.PlayTileAddLetter(b, t.Letter)
|
||||
fb.PlayTileAddBlank(b, t.Blank)
|
||||
offs[i] = fb.PlayTileEnd(b)
|
||||
}
|
||||
fb.EvalRequestStartTilesVector(b, len(offs))
|
||||
for i := len(offs) - 1; i >= 0; i-- {
|
||||
b.PrependUOffsetT(offs[i])
|
||||
}
|
||||
tilesVec := b.EndVector(len(offs))
|
||||
fb.EvalRequestStart(b)
|
||||
fb.EvalRequestAddGameId(b, gid)
|
||||
fb.EvalRequestAddTiles(b, tilesVec)
|
||||
b.Finish(fb.EvalRequestEnd(b))
|
||||
return b.FinishedBytes()
|
||||
}
|
||||
|
||||
// exchange builds an ExchangeRequest payload swapping the listed rack tiles (alphabet
|
||||
// indices; 255 a blank).
|
||||
func exchange(gameID string, tiles []byte) []byte {
|
||||
|
||||
@@ -53,6 +53,15 @@ func (c *Client) Exchange(ctx context.Context, token, gameID string, tiles []byt
|
||||
return decodeMoveResultGame(r.Payload), r.Code, nil
|
||||
}
|
||||
|
||||
// Evaluate previews a tentative play's legality and score without committing it. It is
|
||||
// the per-tile composition call a real client fires (debounced) on every change while
|
||||
// arranging a word, so it is the hottest gameplay request at scale. The harness records
|
||||
// only the result code and latency; an illegal preview is a successful "ok" call.
|
||||
func (c *Client) Evaluate(ctx context.Context, token, gameID string, tiles []PlayTile) (string, error) {
|
||||
r, err := c.execute(ctx, token, msgEvaluate, evalReq(gameID, tiles))
|
||||
return r.Code, err
|
||||
}
|
||||
|
||||
// Nudge prods the opponent whose turn it is.
|
||||
func (c *Client) Nudge(ctx context.Context, token, gameID string) (string, error) {
|
||||
r, err := c.execute(ctx, token, msgNudge, gameAction(gameID))
|
||||
|
||||
@@ -42,19 +42,35 @@ type RealisticConfig struct {
|
||||
GamesPerPlayer int // target concurrent games per player; 0 => random 3..5
|
||||
Tick time.Duration // per-player operation cadence (keeps a player under the per-user limit)
|
||||
SecondaryProb float64 // chance per tick of a non-move operation
|
||||
Eval bool // model the per-tile evaluate preview (the gameplay hot path); false reproduces the pre-evaluate harness
|
||||
EvalRecon int // extra full-composition evaluate re-previews per play, beyond one per placed tile
|
||||
}
|
||||
|
||||
// DefaultRealistic returns the moderate ramp: 50 -> 200
|
||||
// -> 500 concurrent players, ~12 minutes per step, ~1 op/s per player.
|
||||
// -> 500 concurrent players, ~12 minutes per step, ~1 op/s per player, with the
|
||||
// per-tile evaluate preview modelled (the realistic hot path).
|
||||
func DefaultRealistic() RealisticConfig {
|
||||
return RealisticConfig{
|
||||
Steps: []int{50, 200, 500},
|
||||
StepDur: 12 * time.Minute,
|
||||
Tick: 800 * time.Millisecond,
|
||||
SecondaryProb: 0.08,
|
||||
Eval: true,
|
||||
EvalRecon: 1,
|
||||
}
|
||||
}
|
||||
|
||||
// evalGapBase and evalGapSpan bound the modelled pause between successive tile
|
||||
// placements: the client's 250 ms debounce coalesces faster drags into a single
|
||||
// evaluate, so a thoughtful player's previews are spaced by a gap drawn from
|
||||
// [base, base+span] — wide enough that a normal composition stays under the per-user
|
||||
// rate limit, the way a real one does (the limiter's cost is measured by the hammer,
|
||||
// not by self-inflicted rejections here).
|
||||
const (
|
||||
evalGapBase = 250 * time.Millisecond
|
||||
evalGapSpan = 500 * time.Millisecond
|
||||
)
|
||||
|
||||
// RunRealistic runs the staged ramp. Each step activates more players (drawn from the
|
||||
// seeded pool), assembles a cohort of games for them and starts their turn loops; the
|
||||
// loops run until the whole ramp ends. Players from earlier steps keep playing, so
|
||||
@@ -128,7 +144,7 @@ func (d *Driver) playerLoop(ctx context.Context, p seed.Account, games []*Game,
|
||||
d.secondaryOp(ctx, c, p, g, rng)
|
||||
continue
|
||||
}
|
||||
if d.playTurn(ctx, c, p, g, rng) {
|
||||
if d.playTurn(ctx, c, p, g, cfg, rng) {
|
||||
active = slices.DeleteFunc(active, func(x *Game) bool { return x == g })
|
||||
gi = 0
|
||||
if len(active) == 0 {
|
||||
@@ -161,10 +177,10 @@ func (d *Driver) subscribeLoop(ctx context.Context, c *edge.Client, p seed.Accou
|
||||
}
|
||||
|
||||
// playTurn plays one turn in g over the player's client when it is the player's
|
||||
// move: fetch state, replay history, pick a legal move and submit it (or exchange /
|
||||
// pass). It reports whether the game has finished, so the caller can drop it from the
|
||||
// rotation.
|
||||
func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g *Game, rng *rand.Rand) (finished bool) {
|
||||
// move: fetch state, replay history, pick a legal move, compose it (the per-tile
|
||||
// evaluate previews a real client fires) and submit it (or exchange / pass). It reports
|
||||
// whether the game has finished, so the caller can drop it from the rotation.
|
||||
func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g *Game, cfg RealisticConfig, rng *rand.Rand) (finished bool) {
|
||||
seat := g.seatOf(p.ID.String())
|
||||
if seat < 0 {
|
||||
return false
|
||||
@@ -196,6 +212,7 @@ func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g
|
||||
}
|
||||
switch action.Kind {
|
||||
case "play":
|
||||
d.composePlay(ctx, c, p, g, action.Tiles, cfg, rng)
|
||||
t0 = time.Now()
|
||||
_, code, _ := c.SubmitPlay(ctx, p.Token, g.ID, action.Tiles)
|
||||
d.rec.Record("game.submit_play", code, time.Since(t0))
|
||||
@@ -211,6 +228,59 @@ func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g
|
||||
return false
|
||||
}
|
||||
|
||||
// composePlay models a player arranging the chosen play tile by tile before committing:
|
||||
// the debounced evaluate preview the real client fires on each placement (a growing prefix
|
||||
// of the tiles), a few full-composition re-previews for reconsideration (recall a tile, try
|
||||
// another spot), and the single draft persistence the client debounces out. evaluate is the
|
||||
// hottest gameplay request at scale, so omitting it (the pre-evaluate harness) understated
|
||||
// the load; cfg.Eval false reproduces that baseline for an A/B comparison. Every step
|
||||
// honours ctx, so end-of-run cancellation never blocks on a sleep or an in-flight preview.
|
||||
func (d *Driver) composePlay(ctx context.Context, c *edge.Client, p seed.Account, g *Game, tiles []edge.PlayTile, cfg RealisticConfig, rng *rand.Rand) {
|
||||
if !cfg.Eval || len(tiles) == 0 {
|
||||
return
|
||||
}
|
||||
// One evaluate per landed tile: the growing prefix mirrors the client re-previewing
|
||||
// after each placement (an early prefix is often illegal, which is still a successful
|
||||
// "ok" round trip — exactly the backend work a real composition triggers).
|
||||
for n := 1; n <= len(tiles); n++ {
|
||||
if !jitterSleep(ctx, rng, evalGapBase, evalGapSpan) {
|
||||
return
|
||||
}
|
||||
t0 := time.Now()
|
||||
code, _ := c.Evaluate(ctx, p.Token, g.ID, tiles[:n])
|
||||
d.rec.Record("game.evaluate", code, time.Since(t0))
|
||||
}
|
||||
for r := 0; r < cfg.EvalRecon; r++ {
|
||||
if !jitterSleep(ctx, rng, evalGapBase, evalGapSpan) {
|
||||
return
|
||||
}
|
||||
t0 := time.Now()
|
||||
code, _ := c.Evaluate(ctx, p.Token, g.ID, tiles)
|
||||
d.rec.Record("game.evaluate", code, time.Since(t0))
|
||||
}
|
||||
// The client persists the in-progress composition (debounced to one upsert). Its opaque
|
||||
// JSON content does not affect the call's cost, so a minimal valid shape stands in.
|
||||
t0 := time.Now()
|
||||
code, _ := c.DraftSave(ctx, p.Token, g.ID, `{"rack_order":"","board_tiles":[]}`)
|
||||
d.rec.Record("draft.save", code, time.Since(t0))
|
||||
}
|
||||
|
||||
// jitterSleep pauses for a randomised gap in [base, base+span], modelling the human pause
|
||||
// between tile placements that the client's debounce coalesces into one evaluate. It
|
||||
// returns false if ctx is cancelled during the wait, so a composition unwinds promptly at
|
||||
// end of run.
|
||||
func jitterSleep(ctx context.Context, rng *rand.Rand, base, span time.Duration) bool {
|
||||
d := base + time.Duration(rng.Int63n(int64(span)+1))
|
||||
t := time.NewTimer(d)
|
||||
defer t.Stop()
|
||||
select {
|
||||
case <-ctx.Done():
|
||||
return false
|
||||
case <-t.C:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// secondaryOp exercises one of the non-move edge operations the plan calls out, so
|
||||
// the run touches nudge / chat / check-word / draft / profile / stats too, over the
|
||||
// player's own client.
|
||||
|
||||
@@ -29,6 +29,8 @@ import (
|
||||
"go.opentelemetry.io/otel/sdk/resource"
|
||||
sdktrace "go.opentelemetry.io/otel/sdk/trace"
|
||||
"go.opentelemetry.io/otel/trace"
|
||||
|
||||
"scrabble/pkg/version"
|
||||
)
|
||||
|
||||
// Exporter selectors supported per signal.
|
||||
@@ -95,9 +97,7 @@ func New(ctx context.Context, cfg Config) (*Runtime, error) {
|
||||
return nil, err
|
||||
}
|
||||
|
||||
res, err := resource.New(ctx, resource.WithAttributes(
|
||||
attribute.String("service.name", cfg.ServiceName),
|
||||
))
|
||||
res, err := serviceResource(ctx, cfg)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("telemetry: build resource: %w", err)
|
||||
}
|
||||
@@ -122,6 +122,16 @@ func New(ctx context.Context, cfg Config) (*Runtime, error) {
|
||||
return &Runtime{tracerProvider: tracerProvider, meterProvider: meterProvider}, nil
|
||||
}
|
||||
|
||||
// serviceResource builds the OpenTelemetry resource describing this service: its
|
||||
// service.name and the service.version stamped into the binary at build time
|
||||
// (pkg/version, set from the git tag by the deploy).
|
||||
func serviceResource(ctx context.Context, cfg Config) (*resource.Resource, error) {
|
||||
return resource.New(ctx, resource.WithAttributes(
|
||||
attribute.String("service.name", cfg.ServiceName),
|
||||
attribute.String("service.version", version.Version),
|
||||
))
|
||||
}
|
||||
|
||||
// TracerProvider returns the runtime tracer provider, or the global one when r is
|
||||
// not initialised.
|
||||
func (r *Runtime) TracerProvider() trace.TracerProvider {
|
||||
|
||||
@@ -4,6 +4,8 @@ import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"scrabble/pkg/version"
|
||||
)
|
||||
|
||||
// TestConfigValidate covers the supported and rejected exporter selections.
|
||||
@@ -82,3 +84,22 @@ func TestNilRuntime(t *testing.T) {
|
||||
t.Errorf("nil runtime Shutdown: %v", err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestServiceResource checks the resource carries service.name and the embedded
|
||||
// service.version (pkg/version, stamped at build time).
|
||||
func TestServiceResource(t *testing.T) {
|
||||
res, err := serviceResource(context.Background(), DefaultConfig("svc"))
|
||||
if err != nil {
|
||||
t.Fatalf("serviceResource: %v", err)
|
||||
}
|
||||
attrs := map[string]string{}
|
||||
for _, kv := range res.Attributes() {
|
||||
attrs[string(kv.Key)] = kv.Value.AsString()
|
||||
}
|
||||
if attrs["service.name"] != "svc" {
|
||||
t.Errorf("service.name = %q, want svc", attrs["service.name"])
|
||||
}
|
||||
if attrs["service.version"] != version.Version {
|
||||
t.Errorf("service.version = %q, want %q", attrs["service.version"], version.Version)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
// Package version exposes the build version stamped into every Scrabble service
|
||||
// binary. The default is "dev"; release builds override it through the linker
|
||||
// (`go build -ldflags "-X scrabble/pkg/version.Version=<value>"`), wired from the
|
||||
// VERSION build-arg in each service Dockerfile, which the deploy sets to the git
|
||||
// tag (`git describe --tags`). It surfaces as the OpenTelemetry service.version
|
||||
// resource attribute (see pkg/telemetry) and the SPA About screen.
|
||||
package version
|
||||
|
||||
// Version is the build version, "dev" unless overridden at link time.
|
||||
var Version = "dev"
|
||||
@@ -19,8 +19,10 @@ COPY platform/telegram ./platform/telegram
|
||||
# Reduce the workspace to what the platform needs: only pkg + platform/telegram.
|
||||
RUN go work edit -dropuse=./backend -dropuse=./gateway -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0 -dropreplace=scrabble-solver
|
||||
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/validator ./platform/telegram/cmd/validator
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/bot ./platform/telegram/cmd/bot
|
||||
# VERSION (the deploy passes the git tag) is stamped into both binaries via the linker.
|
||||
ARG VERSION=dev
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/validator ./platform/telegram/cmd/validator
|
||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/bot ./platform/telegram/cmd/bot
|
||||
|
||||
# --- validator (home) --------------------------------------------------------
|
||||
FROM gcr.io/distroless/static-debian12:nonroot AS validator
|
||||
|
||||
@@ -191,6 +191,13 @@ func (t *Bot) handleStart(ctx context.Context, api *tgbot.Bot, update *models.Up
|
||||
if update.Message == nil {
|
||||
return
|
||||
}
|
||||
// Reply only in a private chat: the Mini App launch button is an inline web_app
|
||||
// button, which Telegram permits only in private chats — replying to a group message
|
||||
// (the bot is an admin in the moderated chat and now receives its messages) fails with
|
||||
// BUTTON_TYPE_INVALID. In the group the bot only manages permissions, it never chats.
|
||||
if update.Message.Chat.Type != models.ChatTypePrivate {
|
||||
return
|
||||
}
|
||||
startParam := startPayload(update.Message.Text)
|
||||
if _, err := api.SendMessage(ctx, &tgbot.SendMessageParams{
|
||||
ChatID: update.Message.Chat.ID,
|
||||
|
||||
@@ -8,6 +8,7 @@ import (
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/go-telegram/bot/models"
|
||||
"go.uber.org/zap"
|
||||
)
|
||||
|
||||
@@ -103,6 +104,29 @@ func TestTestEnvironmentRoutesGetMe(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleStartRepliesPrivateOnly(t *testing.T) {
|
||||
t.Run("private replies", func(t *testing.T) {
|
||||
api := &fakeBotAPI{}
|
||||
b := newTestBot(t, api)
|
||||
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||
Chat: models.Chat{ID: 42, Type: models.ChatTypePrivate}, Text: "/start g7",
|
||||
}})
|
||||
if api.chatID != "42" || !strings.Contains(api.replyMarkup, "web_app") {
|
||||
t.Errorf("private /start: chat=%q markup=%q, want a web_app reply", api.chatID, api.replyMarkup)
|
||||
}
|
||||
})
|
||||
t.Run("group ignored", func(t *testing.T) {
|
||||
api := &fakeBotAPI{}
|
||||
b := newTestBot(t, api)
|
||||
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||
Chat: models.Chat{ID: -100, Type: models.ChatTypeSupergroup}, Text: "/start",
|
||||
}})
|
||||
if api.chatID != "" {
|
||||
t.Errorf("group /start got a reply (chat=%q); an inline web_app button is invalid in groups", api.chatID)
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func TestStartPayload(t *testing.T) {
|
||||
cases := map[string]string{
|
||||
"/start g123": "g123",
|
||||
|
||||
@@ -92,6 +92,11 @@ func (t *Bot) handleStart(ctx context.Context, api *tgbot.Bot, update *models.Up
|
||||
if update.Message == nil {
|
||||
return
|
||||
}
|
||||
// Only respond to a private /start: the promo bot is a one-on-one onboarding entry
|
||||
// point and should never reply to group messages.
|
||||
if update.Message.Chat.Type != models.ChatTypePrivate {
|
||||
return
|
||||
}
|
||||
if err := t.throttle(ctx); err != nil {
|
||||
return
|
||||
}
|
||||
|
||||
@@ -85,7 +85,7 @@ func TestHandleStartReplies(t *testing.T) {
|
||||
}
|
||||
|
||||
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||
Chat: models.Chat{ID: 42},
|
||||
Chat: models.Chat{ID: 42, Type: models.ChatTypePrivate},
|
||||
From: &models.User{LanguageCode: "ru"},
|
||||
Text: "/start f99",
|
||||
}})
|
||||
@@ -103,3 +103,19 @@ func TestHandleStartReplies(t *testing.T) {
|
||||
t.Errorf("reply_markup = %q, want startapp=f99 (the /start payload forwarded)", api.replyMarkup)
|
||||
}
|
||||
}
|
||||
|
||||
func TestHandleStartIgnoresGroup(t *testing.T) {
|
||||
api := &fakeAPI{}
|
||||
srv := httptest.NewServer(api)
|
||||
t.Cleanup(srv.Close)
|
||||
b, err := New(Config{Token: "1:2", APIBaseURL: srv.URL, BotUsername: "B", BotLinkURL: "https://t.me/b/a"}, zap.NewNop())
|
||||
if err != nil {
|
||||
t.Fatalf("new: %v", err)
|
||||
}
|
||||
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||
Chat: models.Chat{ID: -100, Type: models.ChatTypeSupergroup}, Text: "/start",
|
||||
}})
|
||||
if api.chatID != "" {
|
||||
t.Errorf("replied to a group message (chat=%q); want none", api.chatID)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -238,7 +238,10 @@ test('profile edit disables Save and flags an invalid display name', async ({ pa
|
||||
await expect(save).toBeEnabled();
|
||||
});
|
||||
|
||||
test('link account: a taken email opens the irreversible merge confirmation', async ({ page }) => {
|
||||
// Account linking is hidden in Profile.svelte while we target provider sign-in (the anonymous
|
||||
// /app/ guest who upgrades by linking comes later). The flow is kept wired; re-enable these two
|
||||
// specs together with the `.emailbox` section.
|
||||
test.skip('link account: a taken email opens the irreversible merge confirmation', async ({ page }) => {
|
||||
await loginLobby(page);
|
||||
await openProfile(page);
|
||||
|
||||
@@ -258,7 +261,7 @@ test('link account: a taken email opens the irreversible merge confirmation', as
|
||||
await expect(page.getByText('Merge accounts?')).toBeHidden();
|
||||
});
|
||||
|
||||
test('link account: the Telegram web sign-in control is offered in a browser', async ({ page }) => {
|
||||
test.skip('link account: the Telegram web sign-in control is offered in a browser', async ({ page }) => {
|
||||
await loginLobby(page);
|
||||
await openProfile(page);
|
||||
await expect(page.getByRole('button', { name: 'Link Telegram' })).toBeVisible();
|
||||
|
||||
@@ -83,6 +83,30 @@ test('tg-fullscreen header keeps a constant native-nav gap as the font scales',
|
||||
expect(large.overflows).toBe(false);
|
||||
});
|
||||
|
||||
test('inside Telegram, a failed launch shows the retry screen, not the web login', async ({ page }) => {
|
||||
// initData carrying the mock's "bootfail" sentinel makes authTelegram reject, simulating a
|
||||
// backend outage during launch (e.g. a deploy rolling). The Mini App must surface its own
|
||||
// boot-error/retry screen and never fall back to the web (guest/email) login.
|
||||
await page.addInitScript(() => {
|
||||
Object.assign(window, {
|
||||
Telegram: {
|
||||
WebApp: {
|
||||
initData: 'query_id=bootfail&user=%7B%22id%22%3A1%7D&auth_date=1&hash=deadbeef',
|
||||
initDataUnsafe: {},
|
||||
ready() {},
|
||||
expand() {},
|
||||
},
|
||||
},
|
||||
});
|
||||
});
|
||||
await page.goto('/');
|
||||
|
||||
// After the silent retries, the boot-error screen with its Retry button shows…
|
||||
await expect(page.getByRole('button', { name: 'Retry' })).toBeVisible();
|
||||
// …and the web login (guest) is never shown inside Telegram.
|
||||
await expect(page.getByRole('button', { name: /guest/i })).toHaveCount(0);
|
||||
});
|
||||
|
||||
test('outside Telegram, the /telegram/ entry redirects to the site root', async ({ page }) => {
|
||||
await page.goto('/telegram/');
|
||||
|
||||
|
||||
+6
-1
@@ -18,6 +18,7 @@
|
||||
import CommsHub from './game/CommsHub.svelte';
|
||||
import Feedback from './screens/Feedback.svelte';
|
||||
import Blocked from './screens/Blocked.svelte';
|
||||
import BootError from './screens/BootError.svelte';
|
||||
|
||||
onMount(() => {
|
||||
void bootstrap();
|
||||
@@ -83,6 +84,10 @@
|
||||
{#if !routeIsLobby}
|
||||
<div class="splash">{t('common.loading')}</div>
|
||||
{/if}
|
||||
{:else if app.bootError}
|
||||
<!-- A Mini App launch that failed to authenticate (e.g. the backend was down mid-deploy):
|
||||
show the retry screen instead of falling back to the web login. -->
|
||||
<BootError />
|
||||
{:else if app.blocked}
|
||||
<Blocked />
|
||||
{:else}
|
||||
@@ -123,7 +128,7 @@
|
||||
<StaleInviteModal />
|
||||
<WelcomeRedeemModal />
|
||||
|
||||
{#if routeIsLobby && !app.splashDone && !app.blocked}
|
||||
{#if routeIsLobby && !app.splashDone && !app.blocked && !app.bootError}
|
||||
<Splash />
|
||||
{/if}
|
||||
|
||||
|
||||
@@ -18,6 +18,7 @@ import {
|
||||
telegramDisableVerticalSwipes,
|
||||
telegramHaptic,
|
||||
telegramLaunch,
|
||||
type TelegramLaunch,
|
||||
telegramOnEvent,
|
||||
telegramRequestFullscreen,
|
||||
telegramSetChrome,
|
||||
@@ -41,6 +42,10 @@ export interface Toast {
|
||||
|
||||
export const app = $state<{
|
||||
ready: boolean;
|
||||
/** Inside a Mini App, set when the launch failed to authenticate after its retries (e.g. the
|
||||
* backend was down during a deploy). App.svelte then renders the boot-error retry screen
|
||||
* instead of the web login — a Mini App has no manual sign-in to fall back to. */
|
||||
bootError: boolean;
|
||||
/** Whether the lobby's first cold load has settled (success or error). The loading splash
|
||||
* (components/Splash.svelte) watches it to know when to dismiss; set by screens/Lobby. */
|
||||
lobbyReady: boolean;
|
||||
@@ -90,6 +95,7 @@ export const app = $state<{
|
||||
resync: number;
|
||||
}>({
|
||||
ready: false,
|
||||
bootError: false,
|
||||
lobbyReady: false,
|
||||
splashDone: false,
|
||||
streamAlive: false,
|
||||
@@ -563,14 +569,7 @@ export async function bootstrap(): Promise<void> {
|
||||
// listener above then re-syncs the safe-area insets. Desktop keeps the bot's full-size
|
||||
// window. No-op on clients predating Bot API 8.0.
|
||||
telegramRequestFullscreen();
|
||||
try {
|
||||
await adoptSession(await gateway.authTelegram(launch.initData));
|
||||
// A blocked account skips deep-link routing — the blocked screen overlays every route.
|
||||
if (!app.blocked) await routeStartParam(launch.startParam);
|
||||
} catch (err) {
|
||||
handleError(err);
|
||||
navigate('/login');
|
||||
}
|
||||
await bootTelegram(launch);
|
||||
app.ready = true;
|
||||
return;
|
||||
}
|
||||
@@ -585,6 +584,57 @@ export async function bootstrap(): Promise<void> {
|
||||
app.ready = true;
|
||||
}
|
||||
|
||||
// Inside a Mini App the only identity is the Telegram session, so a failed launch must never fall
|
||||
// back to the web login screen. A transient backend outage (a deploy rolling over) is retried a
|
||||
// few times in silence; only then does the boot-error screen surface, from which Retry re-runs the
|
||||
// same path (retryTelegramBoot).
|
||||
const TELEGRAM_BOOT_RETRIES = 2;
|
||||
const TELEGRAM_BOOT_RETRY_MS = 1200;
|
||||
|
||||
function delay(ms: number): Promise<void> {
|
||||
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||
}
|
||||
|
||||
/**
|
||||
* bootTelegram authenticates a Mini App launch from its initData and routes any deep-link start
|
||||
* parameter, retrying a few times on a transient failure before raising the boot-error screen
|
||||
* (app.bootError). A blocked account is terminal — it switches straight to the blocked screen
|
||||
* without retrying.
|
||||
*/
|
||||
async function bootTelegram(launch: TelegramLaunch): Promise<void> {
|
||||
for (let attempt = 0; ; attempt++) {
|
||||
try {
|
||||
await adoptSession(await gateway.authTelegram(launch.initData));
|
||||
// A blocked account skips deep-link routing — the blocked screen overlays every route.
|
||||
if (!app.blocked) await routeStartParam(launch.startParam);
|
||||
app.bootError = false;
|
||||
return;
|
||||
} catch (err) {
|
||||
if (err instanceof GatewayError && err.code === 'account_blocked') {
|
||||
await enterBlocked();
|
||||
return;
|
||||
}
|
||||
if (attempt >= TELEGRAM_BOOT_RETRIES) {
|
||||
app.bootError = true;
|
||||
return;
|
||||
}
|
||||
await delay(TELEGRAM_BOOT_RETRY_MS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* retryTelegramBoot re-attempts the Mini App launch from the boot-error screen's Retry button. It
|
||||
* clears the error and shows the loading state again, then runs the same retrying boot; on success
|
||||
* the app renders normally, otherwise the boot-error screen returns.
|
||||
*/
|
||||
export async function retryTelegramBoot(): Promise<void> {
|
||||
app.bootError = false;
|
||||
app.ready = false;
|
||||
await bootTelegram(telegramLaunch());
|
||||
app.ready = true;
|
||||
}
|
||||
|
||||
/**
|
||||
* routeStartParam navigates a Telegram deep-link start parameter to its target: a
|
||||
* specific game, the friends screen with a friend-code redemption, or the lobby
|
||||
|
||||
@@ -11,6 +11,9 @@ export const en = {
|
||||
'blocked.temporary': 'Your account is blocked until {until}.',
|
||||
'blocked.reason': 'Reason:',
|
||||
|
||||
'boot.errorTitle': "Couldn't load the game",
|
||||
'boot.errorBody': 'Please try again in a moment.',
|
||||
|
||||
'common.back': 'Back',
|
||||
'common.cancel': 'Cancel',
|
||||
'common.ok': 'OK',
|
||||
|
||||
@@ -12,6 +12,9 @@ export const ru: Record<MessageKey, string> = {
|
||||
'blocked.temporary': 'Ваша учётная запись заблокирована до {until}.',
|
||||
'blocked.reason': 'Причина:',
|
||||
|
||||
'boot.errorTitle': 'Не удалось загрузить игру',
|
||||
'boot.errorBody': 'Попробуйте ещё раз или зайдите позже.',
|
||||
|
||||
'common.back': 'Назад',
|
||||
'common.cancel': 'Отмена',
|
||||
'common.ok': 'ОК',
|
||||
|
||||
@@ -136,7 +136,10 @@ export class MockGateway implements GatewayClient {
|
||||
}
|
||||
|
||||
// --- auth ---
|
||||
async authTelegram(): Promise<Session> {
|
||||
async authTelegram(initData: string): Promise<Session> {
|
||||
// e2e hook: an initData carrying this sentinel simulates a backend that rejects the launch,
|
||||
// so the Mini App boot-failure path (silent retries → boot-error screen) can be exercised.
|
||||
if (initData.includes('bootfail')) throw new GatewayError('unavailable');
|
||||
return { ...SESSION, isGuest: false };
|
||||
}
|
||||
async authGuest(): Promise<Session> {
|
||||
|
||||
@@ -0,0 +1,63 @@
|
||||
<script lang="ts">
|
||||
// The Mini App launch failed to authenticate after its silent retries (e.g. the backend was
|
||||
// briefly down during a deploy). Inside Telegram there is no web login to fall back to, so this
|
||||
// terminal screen offers a manual Retry that re-runs the launch (app.svelte retryTelegramBoot).
|
||||
import { retryTelegramBoot } from '../lib/app.svelte';
|
||||
import { t } from '../lib/i18n/index.svelte';
|
||||
|
||||
let retrying = $state(false);
|
||||
async function retry(): Promise<void> {
|
||||
if (retrying) return;
|
||||
retrying = true;
|
||||
try {
|
||||
await retryTelegramBoot();
|
||||
} finally {
|
||||
retrying = false;
|
||||
}
|
||||
}
|
||||
</script>
|
||||
|
||||
<div class="boot">
|
||||
<div class="card">
|
||||
<h1>{t('boot.errorTitle')}</h1>
|
||||
<p class="msg">{t('boot.errorBody')}</p>
|
||||
<button class="retry" onclick={retry} disabled={retrying}>{t('common.retry')}</button>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<style>
|
||||
.boot {
|
||||
height: 100%;
|
||||
display: grid;
|
||||
place-items: center;
|
||||
padding: 24px;
|
||||
background: var(--bg);
|
||||
}
|
||||
.card {
|
||||
max-width: 28rem;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: center;
|
||||
gap: 1rem;
|
||||
text-align: center;
|
||||
color: var(--text);
|
||||
}
|
||||
h1 {
|
||||
margin: 0;
|
||||
font-size: 1.25rem;
|
||||
}
|
||||
.msg {
|
||||
margin: 0;
|
||||
color: var(--text-muted);
|
||||
}
|
||||
.retry {
|
||||
padding: 9px 16px;
|
||||
border: 1px solid var(--accent);
|
||||
background: var(--accent);
|
||||
color: var(--accent-text);
|
||||
border-radius: var(--radius-sm);
|
||||
}
|
||||
.retry:disabled {
|
||||
opacity: 0.5;
|
||||
}
|
||||
</style>
|
||||
@@ -244,9 +244,10 @@
|
||||
</form>
|
||||
{/if}
|
||||
|
||||
<!-- Linking & merge. Shown to everyone, including guests, who
|
||||
upgrade by binding their first identity. -->
|
||||
<section class="emailbox">
|
||||
<!-- Linking & merge. Hidden for now: we target provider sign-in, and the anonymous
|
||||
/app/ guest (whose upgrade path this is) comes later. Kept wired — drop `hidden`
|
||||
to re-enable, together with the skipped linking specs in e2e/social.spec.ts. -->
|
||||
<section class="emailbox" hidden>
|
||||
<h3>{t('profile.linkAccount')}</h3>
|
||||
{#if !emailSent}
|
||||
<div class="addrow">
|
||||
|
||||
Reference in New Issue
Block a user