Compare commits
18 Commits
fa8abf22db
...
v1.0.0
| Author | SHA1 | Date | |
|---|---|---|---|
| 24017bcb7f | |||
| 40d8f06588 | |||
| c59e522732 | |||
| 8d45ae6e3b | |||
| 2c4f4b10dc | |||
| 520a9092fe | |||
| 9f970495ee | |||
| 3d9ba3ac3d | |||
| 171b71b7e0 | |||
| 2b399d0838 | |||
| f5f45e7afb | |||
| b54cb8878d | |||
| e336638ca8 | |||
| 62f42ed102 | |||
| ecb21bd218 | |||
| e2771826fd | |||
| dec6fac013 | |||
| c494da553a |
@@ -301,8 +301,11 @@ jobs:
|
|||||||
# App version for the About screen: the git tag if present, else the short SHA
|
# App version for the About screen: the git tag if present, else the short SHA
|
||||||
# (the test checkout is shallow/untagged, so this is the SHA here — fine).
|
# (the test checkout is shallow/untagged, so this is the SHA here — fine).
|
||||||
export APP_VERSION="$(git -C "$GITHUB_WORKSPACE" describe --tags --always 2>/dev/null || echo dev)"
|
export APP_VERSION="$(git -C "$GITHUB_WORKSPACE" describe --tags --always 2>/dev/null || echo dev)"
|
||||||
docker compose --ansi never build --progress plain
|
# The telegram-local profile brings the bot + its VPN sidecar; prod runs the
|
||||||
docker compose --ansi never up -d --remove-orphans
|
# bot on its own host instead (deploy/docker-compose.bot.yml), and the prod
|
||||||
|
# main host omits both. Without the profile they would not start here.
|
||||||
|
docker compose --ansi never --profile telegram-local build --progress plain
|
||||||
|
docker compose --ansi never --profile telegram-local up -d --remove-orphans
|
||||||
# The config-only services bind-mount the reseeded config dir. A plain `up -d`
|
# The config-only services bind-mount the reseeded config dir. A plain `up -d`
|
||||||
# leaves them on the previous bind mount (the dir was rm'd + recreated), so a
|
# leaves them on the previous bind mount (the dir was rm'd + recreated), so a
|
||||||
# changed Caddyfile or Grafana dashboard is ignored — force-recreate them to
|
# changed Caddyfile or Grafana dashboard is ignored — force-recreate them to
|
||||||
|
|||||||
@@ -0,0 +1,266 @@
|
|||||||
|
# Manual production rollout. Runs ONLY from master, ONLY on workflow_dispatch with
|
||||||
|
# confirm=deploy (development->master is merged + green first; this is the separate,
|
||||||
|
# deliberate prod step). Visible sequential jobs from most to least significant:
|
||||||
|
# build -> deploy-main -> deploy-bot -> verify
|
||||||
|
# The per-service rolling (postgres->backend->gateway->landing->validator->caddy),
|
||||||
|
# health-gating and auto-rollback live in deploy/prod-deploy.sh on the main host and
|
||||||
|
# show in the deploy-main log. Manual post-deploy rollback is prod-rollback.yaml.
|
||||||
|
# See deploy/README.md (prod runbook).
|
||||||
|
name: prod-deploy
|
||||||
|
run-name: "prod deploy ${{ github.sha }}"
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
confirm:
|
||||||
|
description: 'Type "deploy" to confirm a production rollout from master.'
|
||||||
|
required: true
|
||||||
|
default: ""
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
env:
|
||||||
|
NO_COLOR: "1"
|
||||||
|
DOCKER_CLI_HINTS: "false"
|
||||||
|
REGISTRY: docker.iliadenisov.ru/developer
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
build:
|
||||||
|
if: ${{ github.ref == 'refs/heads/master' && inputs.confirm == 'deploy' }}
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
outputs:
|
||||||
|
tag: ${{ steps.ver.outputs.tag }}
|
||||||
|
env:
|
||||||
|
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||||
|
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||||
|
VITE_TELEGRAM_BOT_ID: ${{ vars.PROD_VITE_TELEGRAM_BOT_ID }}
|
||||||
|
VITE_TELEGRAM_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||||
|
VITE_TELEGRAM_GAME_CHANNEL_NAME: ${{ vars.PROD_VITE_TELEGRAM_GAME_CHANNEL_NAME }}
|
||||||
|
VITE_GATEWAY_URL: ${{ vars.PROD_VITE_GATEWAY_URL }}
|
||||||
|
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||||
|
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||||
|
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
- name: Compute version tag
|
||||||
|
id: ver
|
||||||
|
run: echo "tag=$(git describe --tags --always)" >> "$GITHUB_OUTPUT"
|
||||||
|
- name: Registry login
|
||||||
|
run: echo "$PROD_REGISTRY_PASSWORD" | docker login "${REGISTRY%%/*}" -u "$PROD_REGISTRY_USER" --password-stdin
|
||||||
|
- name: Build and push images
|
||||||
|
working-directory: deploy
|
||||||
|
run: |
|
||||||
|
export TAG="${{ steps.ver.outputs.tag }}" APP_VERSION="${{ steps.ver.outputs.tag }}" SCRABBLE_CONFIG_DIR=.
|
||||||
|
# The four main-stack images via compose (reuses the build args, incl. VERSION);
|
||||||
|
# the bot separately, since it is profiled out of the prod compose.
|
||||||
|
docker compose -f docker-compose.yml -f docker-compose.prod.yml build
|
||||||
|
docker compose -f docker-compose.yml -f docker-compose.prod.yml push backend gateway landing validator
|
||||||
|
docker build -f ../platform/telegram/Dockerfile --target bot --build-arg VERSION="$TAG" -t "$REGISTRY/scrabble-telegram-bot:$TAG" ..
|
||||||
|
docker push "$REGISTRY/scrabble-telegram-bot:$TAG"
|
||||||
|
|
||||||
|
deploy-main:
|
||||||
|
needs: build
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
TAG: ${{ needs.build.outputs.tag }}
|
||||||
|
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||||
|
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||||
|
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||||
|
GRAFANA_ADMIN_PASSWORD: ${{ secrets.PROD_GRAFANA_ADMIN_PASSWORD }}
|
||||||
|
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||||
|
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||||
|
PROD_BOTLINK_GATEWAY_CERT: ${{ secrets.PROD_BOTLINK_GATEWAY_CERT }}
|
||||||
|
PROD_BOTLINK_GATEWAY_KEY: ${{ secrets.PROD_BOTLINK_GATEWAY_KEY }}
|
||||||
|
GM_BASICAUTH_USER: ${{ vars.PROD_GM_BASICAUTH_USER }}
|
||||||
|
GRAFANA_ROOT_URL: ${{ vars.PROD_GRAFANA_ROOT_URL }}
|
||||||
|
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||||
|
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||||
|
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||||
|
POSTGRES_DB: ${{ vars.PROD_POSTGRES_DB }}
|
||||||
|
POSTGRES_USER: ${{ vars.PROD_POSTGRES_USER }}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
with:
|
||||||
|
fetch-depth: 0
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Determine previous tag and migration
|
||||||
|
run: |
|
||||||
|
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||||
|
PREV_TAG="$(ssh_main 'cat /opt/scrabble/DEPLOYED_TAG 2>/dev/null || echo none')"
|
||||||
|
MIGRATION=0
|
||||||
|
if [ "$PREV_TAG" != none ]; then
|
||||||
|
if ! git cat-file -e "$PREV_TAG^{commit}" 2>/dev/null; then
|
||||||
|
MIGRATION=1
|
||||||
|
elif git diff --name-only "$PREV_TAG..$TAG" -- backend/internal/postgres/migrations/ | grep -q .; then
|
||||||
|
MIGRATION=1
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
{ echo "PREV_TAG=$PREV_TAG"; echo "MIGRATION=$MIGRATION"; } >> "$GITHUB_ENV"
|
||||||
|
echo "prev=$PREV_TAG migration=$MIGRATION"
|
||||||
|
- name: Render main env + certs
|
||||||
|
run: |
|
||||||
|
umask 077
|
||||||
|
mkdir -p stage/certs-main
|
||||||
|
cat > stage/env.sh <<EOF
|
||||||
|
export REGISTRY='$REGISTRY'
|
||||||
|
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||||
|
export POSTGRES_DB='${POSTGRES_DB:-scrabble}'
|
||||||
|
export POSTGRES_USER='${POSTGRES_USER:-scrabble}'
|
||||||
|
export POSTGRES_PASSWORD='$POSTGRES_PASSWORD'
|
||||||
|
export GM_BASICAUTH_USER='${GM_BASICAUTH_USER:-gm}'
|
||||||
|
export GM_BASICAUTH_HASH='$GM_BASICAUTH_HASH'
|
||||||
|
export GRAFANA_ADMIN_PASSWORD='$GRAFANA_ADMIN_PASSWORD'
|
||||||
|
export GRAFANA_ROOT_URL='$GRAFANA_ROOT_URL'
|
||||||
|
export CADDY_SITE_ADDRESS='$CADDY_SITE_ADDRESS'
|
||||||
|
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||||
|
export DICT_VERSION='$DICT_VERSION'
|
||||||
|
export APP_VERSION='$TAG'
|
||||||
|
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||||
|
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||||
|
export GATEWAY_ABUSE_BAN_ENABLED='true'
|
||||||
|
EOF
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-main/ca.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_GATEWAY_CERT" > stage/certs-main/gateway.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_GATEWAY_KEY" > stage/certs-main/gateway.key
|
||||||
|
chmod 644 stage/certs-main/*
|
||||||
|
- name: Deploy the main host
|
||||||
|
run: |
|
||||||
|
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||||
|
ssh_main 'mkdir -p /opt/scrabble/compose'
|
||||||
|
tar -C deploy -czf - docker-compose.yml docker-compose.prod.yml prod-deploy.sh \
|
||||||
|
| ssh_main 'tar -C /opt/scrabble/compose -xzf -'
|
||||||
|
tar -C deploy -czf - caddy otelcol prometheus tempo grafana \
|
||||||
|
| ssh_main 'tar -C /opt/scrabble -xzf -'
|
||||||
|
tar -C stage -czf - certs-main \
|
||||||
|
| ssh_main 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||||
|
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.sh "deploy@$MAIN_HOST:/opt/scrabble/env.sh"
|
||||||
|
echo "$PROD_REGISTRY_PASSWORD" | ssh_main "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||||
|
ssh_main "TAG='$TAG' PREV_TAG='$PREV_TAG' MIGRATION='$MIGRATION' bash /opt/scrabble/compose/prod-deploy.sh"
|
||||||
|
|
||||||
|
deploy-bot:
|
||||||
|
needs: [build, deploy-main]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
TAG: ${{ needs.build.outputs.tag }}
|
||||||
|
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||||
|
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
TG_HOST: ${{ vars.PROD_TG_HOST }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||||
|
TELEGRAM_PROMO_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_PROMO_BOT_TOKEN }}
|
||||||
|
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||||
|
PROD_BOTLINK_BOT_CERT: ${{ secrets.PROD_BOTLINK_BOT_CERT }}
|
||||||
|
PROD_BOTLINK_BOT_KEY: ${{ secrets.PROD_BOTLINK_BOT_KEY }}
|
||||||
|
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||||
|
TELEGRAM_GAME_CHANNEL_ID: ${{ vars.PROD_TELEGRAM_GAME_CHANNEL_ID }}
|
||||||
|
TELEGRAM_CHAT_ID: ${{ vars.PROD_TELEGRAM_CHAT_ID }}
|
||||||
|
TELEGRAM_BOT_USERNAME: ${{ vars.PROD_TELEGRAM_BOT_USERNAME }}
|
||||||
|
TELEGRAM_BOT_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Render bot env + certs
|
||||||
|
run: |
|
||||||
|
umask 077
|
||||||
|
mkdir -p stage/certs-bot
|
||||||
|
cat > stage/env.bot.sh <<EOF
|
||||||
|
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||||
|
export BOT_IMAGE='$REGISTRY/scrabble-telegram-bot:$TAG'
|
||||||
|
export BOTLINK_GATEWAY_ADDR='$MAIN_HOST:9443'
|
||||||
|
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||||
|
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||||
|
export TELEGRAM_GAME_CHANNEL_ID='$TELEGRAM_GAME_CHANNEL_ID'
|
||||||
|
export TELEGRAM_CHAT_ID='$TELEGRAM_CHAT_ID'
|
||||||
|
export TELEGRAM_PROMO_BOT_TOKEN='$TELEGRAM_PROMO_BOT_TOKEN'
|
||||||
|
export TELEGRAM_BOT_USERNAME='$TELEGRAM_BOT_USERNAME'
|
||||||
|
export TELEGRAM_BOT_LINK='$TELEGRAM_BOT_LINK'
|
||||||
|
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||||
|
EOF
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-bot/ca.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_BOT_CERT" > stage/certs-bot/bot.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_BOT_KEY" > stage/certs-bot/bot.key
|
||||||
|
chmod 644 stage/certs-bot/*
|
||||||
|
- name: Deploy the bot host
|
||||||
|
run: |
|
||||||
|
ssh_tg() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$TG_HOST" "$@"; }
|
||||||
|
ssh_tg 'mkdir -p /opt/scrabble/compose'
|
||||||
|
tar -C deploy -czf - docker-compose.bot.yml | ssh_tg 'tar -C /opt/scrabble/compose -xzf -'
|
||||||
|
tar -C stage -czf - certs-bot \
|
||||||
|
| ssh_tg 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||||
|
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.bot.sh "deploy@$TG_HOST:/opt/scrabble/env.bot.sh"
|
||||||
|
echo "$PROD_REGISTRY_PASSWORD" | ssh_tg "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||||
|
ssh_tg 'set -a; . /opt/scrabble/env.bot.sh; set +a; cd /opt/scrabble/compose;
|
||||||
|
docker compose -f docker-compose.bot.yml pull;
|
||||||
|
docker compose -f docker-compose.bot.yml up -d'
|
||||||
|
ssh_tg 'for i in $(seq 1 20); do
|
||||||
|
s=$(docker inspect -f "{{.State.Status}}" scrabble-telegram-bot 2>/dev/null || echo missing)
|
||||||
|
r=$(docker inspect -f "{{.State.Restarting}}" scrabble-telegram-bot 2>/dev/null || echo true)
|
||||||
|
if [ "$s" = running ] && [ "$r" = false ]; then
|
||||||
|
c1=$(docker inspect -f "{{.RestartCount}}" scrabble-telegram-bot); sleep 5
|
||||||
|
c2=$(docker inspect -f "{{.RestartCount}}" scrabble-telegram-bot)
|
||||||
|
[ "$c1" = "$c2" ] && { echo "bot healthy"; exit 0; }
|
||||||
|
fi
|
||||||
|
sleep 3
|
||||||
|
done
|
||||||
|
echo "bot not healthy:"; docker logs --tail 80 scrabble-telegram-bot; exit 1'
|
||||||
|
|
||||||
|
verify:
|
||||||
|
needs: [deploy-main, deploy-bot]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||||
|
steps:
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Verify the public site
|
||||||
|
run: |
|
||||||
|
domain="${CADDY_SITE_ADDRESS%% *}"
|
||||||
|
ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "for i in \$(seq 1 20); do
|
||||||
|
if curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/ -o /dev/null &&
|
||||||
|
curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/app/ -o /dev/null &&
|
||||||
|
docker run --rm --network scrabble-internal alpine:3.20 wget -q -T 5 -O /dev/null http://backend:8080/readyz; then
|
||||||
|
echo 'public site + /app/ + backend healthy'; exit 0
|
||||||
|
fi
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
echo 'public verify failed; recent caddy + gateway + backend logs:'
|
||||||
|
docker logs --tail 40 scrabble-caddy; docker logs --tail 40 scrabble-gateway; docker logs --tail 40 scrabble-backend
|
||||||
|
exit 1"
|
||||||
@@ -0,0 +1,223 @@
|
|||||||
|
# Manual production rollback. Runs ONLY from master, ONLY on workflow_dispatch with
|
||||||
|
# confirm=rollback. Re-deploys an already-published image tag (no build): leave
|
||||||
|
# target_version blank to roll back to the previously deployed version (read from the
|
||||||
|
# main host), or set it to a specific release tag from the Releases page. The
|
||||||
|
# re-deploy is the same rolling, health-gated path as prod-deploy (TAG=target,
|
||||||
|
# MIGRATION=0 — rollback is image-only and never migrates the DB; image rollback is
|
||||||
|
# DB-safe under the expand-contract rule). See deploy/README.md (prod runbook).
|
||||||
|
name: prod-rollback
|
||||||
|
run-name: "prod rollback ${{ inputs.target_version || 'previous' }}"
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
inputs:
|
||||||
|
confirm:
|
||||||
|
description: 'Type "rollback" to confirm a production rollback.'
|
||||||
|
required: true
|
||||||
|
default: ""
|
||||||
|
target_version:
|
||||||
|
description: "Release tag to roll back to (blank = the previous deployed version)."
|
||||||
|
required: false
|
||||||
|
default: ""
|
||||||
|
|
||||||
|
permissions:
|
||||||
|
contents: read
|
||||||
|
|
||||||
|
env:
|
||||||
|
NO_COLOR: "1"
|
||||||
|
DOCKER_CLI_HINTS: "false"
|
||||||
|
REGISTRY: docker.iliadenisov.ru/developer
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
rollback-main:
|
||||||
|
if: ${{ github.ref == 'refs/heads/master' && inputs.confirm == 'rollback' }}
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
outputs:
|
||||||
|
target: ${{ steps.resolve.outputs.target }}
|
||||||
|
env:
|
||||||
|
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||||
|
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
POSTGRES_PASSWORD: ${{ secrets.PROD_POSTGRES_PASSWORD }}
|
||||||
|
GM_BASICAUTH_HASH: ${{ secrets.PROD_GM_BASICAUTH_HASH }}
|
||||||
|
GRAFANA_ADMIN_PASSWORD: ${{ secrets.PROD_GRAFANA_ADMIN_PASSWORD }}
|
||||||
|
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||||
|
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||||
|
PROD_BOTLINK_GATEWAY_CERT: ${{ secrets.PROD_BOTLINK_GATEWAY_CERT }}
|
||||||
|
PROD_BOTLINK_GATEWAY_KEY: ${{ secrets.PROD_BOTLINK_GATEWAY_KEY }}
|
||||||
|
GM_BASICAUTH_USER: ${{ vars.PROD_GM_BASICAUTH_USER }}
|
||||||
|
GRAFANA_ROOT_URL: ${{ vars.PROD_GRAFANA_ROOT_URL }}
|
||||||
|
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||||
|
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||||
|
DICT_VERSION: ${{ vars.PROD_DICT_VERSION }}
|
||||||
|
POSTGRES_DB: ${{ vars.PROD_POSTGRES_DB }}
|
||||||
|
POSTGRES_USER: ${{ vars.PROD_POSTGRES_USER }}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||||
|
INPUT_TARGET: ${{ inputs.target_version }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Resolve rollback target
|
||||||
|
id: resolve
|
||||||
|
run: |
|
||||||
|
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||||
|
CURRENT="$(ssh_main 'cat /opt/scrabble/DEPLOYED_TAG 2>/dev/null || echo none')"
|
||||||
|
if [ -n "$INPUT_TARGET" ]; then
|
||||||
|
TARGET="$INPUT_TARGET"
|
||||||
|
else
|
||||||
|
TARGET="$(ssh_main 'cat /opt/scrabble/PREVIOUS_TAG 2>/dev/null || echo none')"
|
||||||
|
fi
|
||||||
|
if [ -z "$TARGET" ] || [ "$TARGET" = none ]; then
|
||||||
|
echo "no rollback target (no PREVIOUS_TAG on the host and no target_version input)"; exit 1
|
||||||
|
fi
|
||||||
|
if [ "$TARGET" = "$CURRENT" ]; then
|
||||||
|
echo "target $TARGET is already the deployed version; nothing to do"; exit 1
|
||||||
|
fi
|
||||||
|
echo "rolling back: current=$CURRENT -> target=$TARGET"
|
||||||
|
echo "target=$TARGET" >> "$GITHUB_OUTPUT"
|
||||||
|
{ echo "TARGET=$TARGET"; echo "CURRENT=$CURRENT"; } >> "$GITHUB_ENV"
|
||||||
|
- name: Render main env + certs
|
||||||
|
run: |
|
||||||
|
umask 077
|
||||||
|
mkdir -p stage/certs-main
|
||||||
|
cat > stage/env.sh <<EOF
|
||||||
|
export REGISTRY='$REGISTRY'
|
||||||
|
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||||
|
export POSTGRES_DB='${POSTGRES_DB:-scrabble}'
|
||||||
|
export POSTGRES_USER='${POSTGRES_USER:-scrabble}'
|
||||||
|
export POSTGRES_PASSWORD='$POSTGRES_PASSWORD'
|
||||||
|
export GM_BASICAUTH_USER='${GM_BASICAUTH_USER:-gm}'
|
||||||
|
export GM_BASICAUTH_HASH='$GM_BASICAUTH_HASH'
|
||||||
|
export GRAFANA_ADMIN_PASSWORD='$GRAFANA_ADMIN_PASSWORD'
|
||||||
|
export GRAFANA_ROOT_URL='$GRAFANA_ROOT_URL'
|
||||||
|
export CADDY_SITE_ADDRESS='$CADDY_SITE_ADDRESS'
|
||||||
|
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||||
|
export DICT_VERSION='$DICT_VERSION'
|
||||||
|
export APP_VERSION='$TARGET'
|
||||||
|
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||||
|
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||||
|
export GATEWAY_ABUSE_BAN_ENABLED='true'
|
||||||
|
EOF
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-main/ca.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_GATEWAY_CERT" > stage/certs-main/gateway.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_GATEWAY_KEY" > stage/certs-main/gateway.key
|
||||||
|
chmod 644 stage/certs-main/*
|
||||||
|
- name: Roll the main host back
|
||||||
|
run: |
|
||||||
|
ssh_main() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "$@"; }
|
||||||
|
ssh_main 'mkdir -p /opt/scrabble/compose'
|
||||||
|
tar -C deploy -czf - docker-compose.yml docker-compose.prod.yml prod-deploy.sh \
|
||||||
|
| ssh_main 'tar -C /opt/scrabble/compose -xzf -'
|
||||||
|
tar -C deploy -czf - caddy otelcol prometheus tempo grafana \
|
||||||
|
| ssh_main 'tar -C /opt/scrabble -xzf -'
|
||||||
|
tar -C stage -czf - certs-main \
|
||||||
|
| ssh_main 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||||
|
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.sh "deploy@$MAIN_HOST:/opt/scrabble/env.sh"
|
||||||
|
echo "$PROD_REGISTRY_PASSWORD" | ssh_main "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||||
|
# Image-only rollback: no migration window (TAG=target, MIGRATION=0). A failed
|
||||||
|
# rollback's auto-revert returns to the current version (PREV_TAG=$CURRENT).
|
||||||
|
ssh_main "TAG='$TARGET' PREV_TAG='$CURRENT' MIGRATION=0 bash /opt/scrabble/compose/prod-deploy.sh"
|
||||||
|
|
||||||
|
rollback-bot:
|
||||||
|
needs: rollback-main
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
TARGET: ${{ needs.rollback-main.outputs.target }}
|
||||||
|
PROD_REGISTRY_USER: ${{ vars.PROD_REGISTRY_USER }}
|
||||||
|
PROD_REGISTRY_PASSWORD: ${{ secrets.PROD_REGISTRY_PASSWORD }}
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
TG_HOST: ${{ vars.PROD_TG_HOST }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
TELEGRAM_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_BOT_TOKEN }}
|
||||||
|
TELEGRAM_PROMO_BOT_TOKEN: ${{ secrets.PROD_TELEGRAM_PROMO_BOT_TOKEN }}
|
||||||
|
PROD_BOTLINK_CA: ${{ secrets.PROD_BOTLINK_CA }}
|
||||||
|
PROD_BOTLINK_BOT_CERT: ${{ secrets.PROD_BOTLINK_BOT_CERT }}
|
||||||
|
PROD_BOTLINK_BOT_KEY: ${{ secrets.PROD_BOTLINK_BOT_KEY }}
|
||||||
|
LOG_LEVEL: ${{ vars.PROD_LOG_LEVEL }}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${{ vars.PROD_TELEGRAM_MINIAPP_URL }}
|
||||||
|
TELEGRAM_GAME_CHANNEL_ID: ${{ vars.PROD_TELEGRAM_GAME_CHANNEL_ID }}
|
||||||
|
TELEGRAM_CHAT_ID: ${{ vars.PROD_TELEGRAM_CHAT_ID }}
|
||||||
|
TELEGRAM_BOT_USERNAME: ${{ vars.PROD_TELEGRAM_BOT_USERNAME }}
|
||||||
|
TELEGRAM_BOT_LINK: ${{ vars.PROD_VITE_TELEGRAM_LINK }}
|
||||||
|
steps:
|
||||||
|
- uses: actions/checkout@v4
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Render bot env + certs
|
||||||
|
run: |
|
||||||
|
umask 077
|
||||||
|
mkdir -p stage/certs-bot
|
||||||
|
cat > stage/env.bot.sh <<EOF
|
||||||
|
export SCRABBLE_CONFIG_DIR='/opt/scrabble'
|
||||||
|
export BOT_IMAGE='$REGISTRY/scrabble-telegram-bot:$TARGET'
|
||||||
|
export BOTLINK_GATEWAY_ADDR='$MAIN_HOST:9443'
|
||||||
|
export TELEGRAM_BOT_TOKEN='$TELEGRAM_BOT_TOKEN'
|
||||||
|
export TELEGRAM_MINIAPP_URL='$TELEGRAM_MINIAPP_URL'
|
||||||
|
export TELEGRAM_GAME_CHANNEL_ID='$TELEGRAM_GAME_CHANNEL_ID'
|
||||||
|
export TELEGRAM_CHAT_ID='$TELEGRAM_CHAT_ID'
|
||||||
|
export TELEGRAM_PROMO_BOT_TOKEN='$TELEGRAM_PROMO_BOT_TOKEN'
|
||||||
|
export TELEGRAM_BOT_USERNAME='$TELEGRAM_BOT_USERNAME'
|
||||||
|
export TELEGRAM_BOT_LINK='$TELEGRAM_BOT_LINK'
|
||||||
|
export LOG_LEVEL='${LOG_LEVEL:-info}'
|
||||||
|
EOF
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_CA" > stage/certs-bot/ca.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_BOT_CERT" > stage/certs-bot/bot.crt
|
||||||
|
printf '%s\n' "$PROD_BOTLINK_BOT_KEY" > stage/certs-bot/bot.key
|
||||||
|
chmod 644 stage/certs-bot/*
|
||||||
|
- name: Roll the bot host back
|
||||||
|
run: |
|
||||||
|
ssh_tg() { ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$TG_HOST" "$@"; }
|
||||||
|
ssh_tg 'mkdir -p /opt/scrabble/compose'
|
||||||
|
tar -C deploy -czf - docker-compose.bot.yml | ssh_tg 'tar -C /opt/scrabble/compose -xzf -'
|
||||||
|
tar -C stage -czf - certs-bot \
|
||||||
|
| ssh_tg 'rm -rf /opt/scrabble/certs && mkdir -p /opt/scrabble/certs && tar -C /opt/scrabble/certs --strip-components=1 -xzf -'
|
||||||
|
scp -i ~/.ssh/id_deploy -o BatchMode=yes stage/env.bot.sh "deploy@$TG_HOST:/opt/scrabble/env.bot.sh"
|
||||||
|
echo "$PROD_REGISTRY_PASSWORD" | ssh_tg "docker login ${REGISTRY%%/*} -u $PROD_REGISTRY_USER --password-stdin"
|
||||||
|
ssh_tg 'set -a; . /opt/scrabble/env.bot.sh; set +a; cd /opt/scrabble/compose;
|
||||||
|
docker compose -f docker-compose.bot.yml pull;
|
||||||
|
docker compose -f docker-compose.bot.yml up -d'
|
||||||
|
|
||||||
|
verify:
|
||||||
|
needs: [rollback-main, rollback-bot]
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
defaults:
|
||||||
|
run:
|
||||||
|
shell: bash
|
||||||
|
env:
|
||||||
|
PROD_SSH_KEY: ${{ secrets.PROD_SSH_KEY }}
|
||||||
|
PROD_SSH_KNOWN_HOSTS: ${{ secrets.PROD_SSH_KNOWN_HOSTS }}
|
||||||
|
MAIN_HOST: ${{ vars.PROD_MAIN_HOST }}
|
||||||
|
CADDY_SITE_ADDRESS: ${{ vars.PROD_CADDY_SITE_ADDRESS }}
|
||||||
|
steps:
|
||||||
|
- name: Set up SSH
|
||||||
|
run: |
|
||||||
|
mkdir -p ~/.ssh && chmod 700 ~/.ssh
|
||||||
|
printf '%s\n' "$PROD_SSH_KEY" > ~/.ssh/id_deploy && chmod 600 ~/.ssh/id_deploy
|
||||||
|
printf '%s\n' "$PROD_SSH_KNOWN_HOSTS" > ~/.ssh/known_hosts
|
||||||
|
- name: Verify the public site
|
||||||
|
run: |
|
||||||
|
domain="${CADDY_SITE_ADDRESS%% *}"
|
||||||
|
ssh -i ~/.ssh/id_deploy -o BatchMode=yes "deploy@$MAIN_HOST" "for i in \$(seq 1 20); do
|
||||||
|
if curl -fsS -k --resolve $domain:443:127.0.0.1 https://$domain/ -o /dev/null &&
|
||||||
|
docker run --rm --network scrabble-internal alpine:3.20 wget -q -T 5 -O /dev/null http://backend:8080/readyz; then
|
||||||
|
echo 'rolled-back site healthy'; exit 0
|
||||||
|
fi
|
||||||
|
sleep 5
|
||||||
|
done
|
||||||
|
echo 'verify failed'; docker logs --tail 40 scrabble-caddy; docker logs --tail 40 scrabble-backend; exit 1"
|
||||||
@@ -51,7 +51,7 @@ independent (see ARCHITECTURE §9.1).
|
|||||||
| 15 | Dual Telegram bots & language-gated variants | **done** |
|
| 15 | Dual Telegram bots & language-gated variants | **done** |
|
||||||
| 16 | Deploy infra & test contour (Dockerfiles, gateway static UI, compose, observability) | **done** |
|
| 16 | Deploy infra & test contour (Dockerfiles, gateway static UI, compose, observability) | **done** |
|
||||||
| 17 | Test-contour verification & defect fixes | **done** |
|
| 17 | Test-contour verification & defect fixes | **done** |
|
||||||
| 18 | Prod contour deploy (SSH export/import, manual after merge) | todo |
|
| 18 | Prod contour deploy (registry, two-host, rolling + auto-rollback; manual after merge) | machinery built; first cutover pending DNS |
|
||||||
| 19 | User feedback (in-app submit + attachment, admin review/reply, account roles) | **done** |
|
| 19 | User feedback (in-app submit + attachment, admin review/reply, account roles) | **done** |
|
||||||
|
|
||||||
Scaffolding is incremental: `go.work` lists only existing modules; each stage
|
Scaffolding is incremental: `go.work` lists only existing modules; each stage
|
||||||
@@ -413,18 +413,28 @@ raw list is kept here as the record of what the first contour run surfaced.
|
|||||||
"что-то пошло не так". при этом "new -> эрудит" работает. Попробуй посмотреть в логах сейчас, может что-то есть. Или как-то иначе проанализируй, или давай вместе будем смотреть, если не получится.
|
"что-то пошло не так". при этом "new -> эрудит" работает. Попробуй посмотреть в логах сейчас, может что-то есть. Или как-то иначе проанализируй, или давай вместе будем смотреть, если не получится.
|
||||||
|
|
||||||
### Stage 18 — Prod contour deploy
|
### Stage 18 — Prod contour deploy
|
||||||
Scope: the **production contour** on a remote host over SSH. Deploy by **container export/import**
|
Scope: the **production contour** on **two remote hosts** over SSH — main (full stack, `erudit-game.ru`)
|
||||||
(`docker save` → `scp`/ssh → `docker load` → `docker compose up` on the remote), the SSH key + host IP
|
and tg (the bot only). Resolved open details (re-interviewed):
|
||||||
in Gitea secrets; **strictly manual** (`workflow_dispatch`) after `development` is merged to `master`
|
- **Transport: a registry** (not export/import) — build + push to `docker.iliadenisov.ru`, the hosts pull by tag.
|
||||||
(the Stage 16 branch model: `feature/* → development → master`, merge gated green). Two-contour config
|
- **Cert: ACME** at the contour caddy (`CADDY_SITE_ADDRESS=erudit-game.ru www.erudit-game.ru`, no host caddy).
|
||||||
uses **`TEST_`/`PROD_` secret/variable prefixes** — Gitea 1.26 has no deployment environments (verified:
|
- **No prod VPN** — the bot host has native Bot API egress (verified `api.telegram.org` → 200).
|
||||||
the `environments` API 404s), so a flat prefixed namespace is the convention.
|
- **Rollback** — rolling per-service deploy (least → most dependent), health-gated, auto-rollback to the
|
||||||
Reuses the Stage 16 `deploy/docker-compose.yml` as-is, mapping the **`PROD_`** set onto the same
|
previous image tag; a maintenance window + consistent `pg_dump` only on a schema migration
|
||||||
unprefixed compose vars. **No host caddy on prod**, so the contour's own caddy terminates TLS — set
|
(expand-contract keeps the auto-rollback image-only; the dump is a manual safety net).
|
||||||
`CADDY_SITE_ADDRESS` to the prod domain so caddy does its own ACME (the Caddyfile is already
|
|
||||||
parameterised for this; the test contour leaves it `:80` behind the host caddy).
|
**Strictly manual** (`workflow_dispatch` from `master`, `confirm=deploy`) after `development → master`
|
||||||
Open details (re-interview): export/import vs a registry trade-off; prod domain/cert source (ACME vs a
|
is merged green. `TEST_`/`PROD_` prefixed Gitea secrets/variables (Gitea 1.26 has no deployment
|
||||||
provided cert) at the contour caddy; prod VPN; rollback.
|
environments — the `environments` API 404s). Hosts are provisioned by **`deploy/ansible/`** (docker, a
|
||||||
|
non-sudo `deploy` user with the CI key, key-only sshd, ufw, fail2ban). The main host is **launch-sized**
|
||||||
|
(2 vCPU / 1.9 GiB): `docker-compose.prod.yml` trims the R7 limits (`GOMAXPROCS=2`, smaller caps, 7d
|
||||||
|
Prometheus retention) and adds `node_exporter` for host-memory monitoring (launch undersized, resize at
|
||||||
|
Selectel reactively). `vpn`+`bot` are gated to a `telegram-local` compose profile (test only); the prod
|
||||||
|
bot runs standalone from `docker-compose.bot.yml`. `GATEWAY_ABUSE_BAN_ENABLED=true`.
|
||||||
|
|
||||||
|
**Built:** `deploy/ansible/` (both hosts provisioned + verified), the compose split + `node_exporter`,
|
||||||
|
`.gitea/workflows/prod-deploy.yaml` + `deploy/prod-deploy.sh`, the full `PROD_` secret/variable set.
|
||||||
|
**Remaining (acceptance):** the **first live cutover** — waits on the `erudit-game.ru` DNS delegation
|
||||||
|
(`A`/`www` → the main host) that ACME requires; then run the workflow and verify the public site end-to-end.
|
||||||
|
|
||||||
### Stage 19 — User feedback *(done)*
|
### Stage 19 — User feedback *(done)*
|
||||||
A user→operator feedback channel, sequenced after the numbered stages but shipped **before** the Stage 18
|
A user→operator feedback channel, sequenced after the numbered stages but shipped **before** the Stage 18
|
||||||
|
|||||||
+10
-5
@@ -39,8 +39,8 @@ the edge before prod. Each phase maps back to the owner's raw pre-release TODO l
|
|||||||
| FM | First-move tile draw (official rules): each seated player draws a tile, the one closest to "A" leads (a blank beats every letter), ties re-drawing until a single leader; **honest per-draw `crypto/rand` entropy**, not the bag seed, so the **record** (`game_setup_draws`, migration `00013`) — not a seed — is the only account of the outcome, kept for future **tournaments** (designed as a discrete per-tile "player N draws" step). Friend/AI draws at create; **auto-match draws at *open*** against a synthetic `uuid.Nil` opponent whose draw rows are back-filled on join, so the opener's seat is fixed up front and the existing open-game pre-move is preserved (no reseating, no play-gating). Admin `/_gm/games/:id` gains the recorded draw list + a simple **step-by-step board replay** (`ReplayTimeline`). | owner ad-hoc | **done** |
|
| FM | First-move tile draw (official rules): each seated player draws a tile, the one closest to "A" leads (a blank beats every letter), ties re-drawing until a single leader; **honest per-draw `crypto/rand` entropy**, not the bag seed, so the **record** (`game_setup_draws`, migration `00013`) — not a seed — is the only account of the outcome, kept for future **tournaments** (designed as a discrete per-tile "player N draws" step). Friend/AI draws at create; **auto-match draws at *open*** against a synthetic `uuid.Nil` opponent whose draw rows are back-filled on join, so the opener's seat is fixed up front and the existing open-game pre-move is preserved (no reseating, no play-gating). Admin `/_gm/games/:id` gains the recorded draw list + a simple **step-by-step board replay** (`ReplayTimeline`). | owner ad-hoc | **done** |
|
||||||
| SB | Single Telegram bot + per-user variant preferences: the two per-language bots collapse into **one** (drop `accounts.service_language`, `supported_languages`, the `*_EN`/`*_RU` env vars and game-language push routing — the single bot renders in the recipient's `preferred_language`); New Game variant gating moves to a profile **`variant_preferences`** set (default Erudit only, Erudit-first, server-enforced on the caller's auto-match/vs-AI/invitation-create paths, an invited friend may accept any variant); env vars collapse to unsuffixed `TELEGRAM_BOT_TOKEN`/`TELEGRAM_GAME_CHANNEL_ID`/`VITE_TELEGRAM_LINK`/`VITE_TELEGRAM_GAME_CHANNEL_NAME` and `GATEWAY_DEFAULT_SUPPORTED_LANGUAGES` is removed; wire drops `service_language`/`supported_languages` (Session, ValidateInitDataResponse) + the push `language` routing field and adds `variant_preferences` to Profile/UpdateProfile. | owner ad-hoc | **done** |
|
| SB | Single Telegram bot + per-user variant preferences: the two per-language bots collapse into **one** (drop `accounts.service_language`, `supported_languages`, the `*_EN`/`*_RU` env vars and game-language push routing — the single bot renders in the recipient's `preferred_language`); New Game variant gating moves to a profile **`variant_preferences`** set (default Erudit only, Erudit-first, server-enforced on the caller's auto-match/vs-AI/invitation-create paths, an invited friend may accept any variant); env vars collapse to unsuffixed `TELEGRAM_BOT_TOKEN`/`TELEGRAM_GAME_CHANNEL_ID`/`VITE_TELEGRAM_LINK`/`VITE_TELEGRAM_GAME_CHANNEL_NAME` and `GATEWAY_DEFAULT_SUPPORTED_LANGUAGES` is removed; wire drops `service_language`/`supported_languages` (Session, ValidateInitDataResponse) + the push `language` routing field and adds `variant_preferences` to Profile/UpdateProfile. | owner ad-hoc | **done** |
|
||||||
| DV | Dictionary version hygiene: CI + image/compose seed track the current release (`v1.2.1`); a **seed-drift guard** records the flat dir's seed in an authoritative `.seed_version` marker so a bumped build seed on a live volume is ignored (it can't relabel live bytes — which would mis-serve the dictionary + void games pinned to the prior label); `DICT_VERSION` is the fresh-volume seed only, a live contour migrates through the admin console | owner ad-hoc | **done** |
|
| DV | Dictionary version hygiene: CI + image/compose seed track the current release (`v1.2.1`); a **seed-drift guard** records the flat dir's seed in an authoritative `.seed_version` marker so a bumped build seed on a live volume is ignored (it can't relabel live bytes — which would mis-serve the dictionary + void games pinned to the prior label); `DICT_VERSION` is the fresh-volume seed only, a live contour migrates through the admin console | owner ad-hoc | **done** |
|
||||||
| TX | Telegram egress off the main host: split the connector into a home **validator** (Mini App / Login-Widget HMAC, no VPN, no Bot API — so game login no longer depends on Telegram being reachable) and a remote **bot** (Bot API long-poll + `sendMessage`) that holds **no inbound port** and dials the gateway over a reverse **mTLS bot-link** (`pkg/proto/botlink/v1`); the gateway funnels out-of-app push (fire-and-forget, at-most-once) and the backend admin broadcasts (a relay that awaits the bot's ack) down the link. The bot is Telegram-rate-limited; **one bot now**, with seams (a bot registry + `owns_updates` + command ids) for N later; **no webhook** (rejected: one URL per token, adds inbound + a static address). The **unified test contour** runs the split (the bot keeps its VPN sidecar and dials the gateway by its internal name; certs from `deploy/gen-certs.sh`). The **prod** wiring — the bot on a separate host (no VPN), the gateway bot-link port published, `PROD_` certs with scheduled rotation, an SSH deploy of both hosts together — is the **deferred final stage** (Stage 18). | owner ad-hoc | **done** (code + test contour; prod wiring → Stage 18) |
|
| TX | Telegram egress off the main host: split the connector into a home **validator** (Mini App / Login-Widget HMAC, no VPN, no Bot API — so game login no longer depends on Telegram being reachable) and a remote **bot** (Bot API long-poll + `sendMessage`) that holds **no inbound port** and dials the gateway over a reverse **mTLS bot-link** (`pkg/proto/botlink/v1`); the gateway funnels out-of-app push (fire-and-forget, at-most-once) and the backend admin broadcasts (a relay that awaits the bot's ack) down the link. The bot is Telegram-rate-limited; **one bot now**, with seams (a bot registry + `owns_updates` + command ids) for N later; **no webhook** (rejected: one URL per token, adds inbound + a static address). The **unified test contour** runs the split (the bot keeps its VPN sidecar and dials the gateway by its internal name; certs from `deploy/gen-certs.sh`). The **prod** wiring — the bot on a separate host (no VPN), the gateway bot-link port published, `PROD_` certs, an SSH deploy of both hosts together — is **built in Stage 18** (the two-host registry rollout; first cutover pending the `erudit-game.ru` DNS). | owner ad-hoc | **done** (code + test contour; prod wiring built — Stage 18) |
|
||||||
| AG | Anti-abuse IP ban + honeypot/honeytoken (prod-only): a fail2ban-style in-memory `ratelimit.Banlist` keyed by client IP, fed by sustained rate-limiter rejections (the IP-keyed public/email/admin classes — the user class stays the soft-flag's concern), a **honeypot** decoy path (the contour caddy tags `/.env`, `/.git`, `/wp-*`, … with `X-Scrabble-Honeypot` and routes them to the gateway), and a **honeytoken** (`GATEWAY_HONEYTOKEN`, a planted bearer). The `abuseGuard` edge middleware refuses a banned IP with **429** before any work — closing the R3 gap that the static SPA/landing was outside the token bucket. Off by default — it keys by the real client IP the shared-NAT test contour does not expose (detection still logs there); enabled in prod via `GATEWAY_ABUSE_BAN_ENABLED`. Operators see + lift bans on the console **Throttled** page; the gateway syncs its active set to the backend (`/api/v1/internal/bans/sync`, `internal/banview`) every 30 s and applies operator unbans. | owner ad-hoc | **done** (code + test contour; ban enabled in prod → Stage 18) |
|
| AG | Anti-abuse IP ban + honeypot/honeytoken (prod-only): a fail2ban-style in-memory `ratelimit.Banlist` keyed by client IP, fed by sustained rate-limiter rejections (the IP-keyed public/email/admin classes — the user class stays the soft-flag's concern), a **honeypot** decoy path (the contour caddy tags `/.env`, `/.git`, `/wp-*`, … with `X-Scrabble-Honeypot` and routes them to the gateway), and a **honeytoken** (`GATEWAY_HONEYTOKEN`, a planted bearer). The `abuseGuard` edge middleware refuses a banned IP with **429** before any work — closing the R3 gap that the static SPA/landing was outside the token bucket. Off by default — it keys by the real client IP the shared-NAT test contour does not expose (detection still logs there); enabled in prod via `GATEWAY_ABUSE_BAN_ENABLED`. Operators see + lift bans on the console **Throttled** page; the gateway syncs its active set to the backend (`/api/v1/internal/bans/sync`, `internal/banview`) every 30 s and applies operator unbans. | owner ad-hoc | **done** (code + test contour; ban on in prod via Stage 18 — machinery built, cutover pending DNS) |
|
||||||
| CM | Channel-chat moderation + promo bot: a second standalone bot in the bot container answers `/start` with a localized message + a **URL** button into the **main** bot's Mini App (`?startapp`; a `web_app` button would sign initData with the promo token, which the main validator rejects). The **main** bot gates write access in a channel's linked discussion chat. The chat **allows sending by default** and the bot only restricts (Telegram intersects the chat default with the per-user permission, so a per-user grant cannot exceed a deny-by-default group): it **mutes** a member who is not registered or is admin-suspended or holding a new **`chat_muted`** role, and **un-mutes** an eligible one it had muted, for a member currently in the chat (a `getChatMember` guard, since bots cannot list members). Eligibility = `registered AND NOT suspended AND NOT chat_muted` (the game suspension dominates), resolved once in the backend and reached two ways: the bot's `ResolveChatEligibility` on a `chat_member` event over the existing mTLS bot-link, and a backend `chat_access_changed` event → gateway → `ChatGate` command (emitted on block/unblock, a `chat_muted` change, a first registration, or a temporary-block expiry via a sweeper; idempotent). No schema change — `chat_muted` reuses `account_roles`. | owner ad-hoc | **done** |
|
| CM | Channel-chat moderation + promo bot: a second standalone bot in the bot container answers `/start` with a localized message + a **URL** button into the **main** bot's Mini App (`?startapp`; a `web_app` button would sign initData with the promo token, which the main validator rejects). The **main** bot gates write access in a channel's linked discussion chat. The chat **allows sending by default** and the bot only restricts (Telegram intersects the chat default with the per-user permission, so a per-user grant cannot exceed a deny-by-default group): it **mutes** a member who is not registered or is admin-suspended or holding a new **`chat_muted`** role, and **un-mutes** an eligible one it had muted, for a member currently in the chat (a `getChatMember` guard, since bots cannot list members). Eligibility = `registered AND NOT suspended AND NOT chat_muted` (the game suspension dominates), resolved once in the backend and reached two ways: the bot's `ResolveChatEligibility` on a `chat_member` event over the existing mTLS bot-link, and a backend `chat_access_changed` event → gateway → `ChatGate` command (emitted on block/unblock, a `chat_muted` change, a first registration, or a temporary-block expiry via a sweeper; idempotent). No schema change — `chat_muted` reuses `account_roles`. | owner ad-hoc | **done** |
|
||||||
| → | Stage 18 — prod contour deploy | — | see [`PLAN.md`](PLAN.md) |
|
| → | Stage 18 — prod contour deploy | — | see [`PLAN.md`](PLAN.md) |
|
||||||
|
|
||||||
@@ -311,7 +311,7 @@ Then Stage 18.
|
|||||||
hammer (99.97 % rejected, p99 2 ms). **Top finding:** ~14 % `transport_error` on `game.state` at 500
|
hammer (99.97 % rejected, p99 2 ms). **Top finding:** ~14 % `transport_error` on `game.state` at 500
|
||||||
players, under CPU saturation (backend/gateway/Postgres each ~1 core) and amplified by the harness's
|
players, under CPU saturation (backend/gateway/Postgres each ~1 core) and amplified by the harness's
|
||||||
single shared `http2.Transport`; the harness itself peaked at 86 % of a core on the same host, so the
|
single shared `http2.Transport`; the harness itself peaked at 86 % of a core on the same host, so the
|
||||||
figures are pessimistic. Full trip report in [`../loadtest/REPORT-R2.md`](../loadtest/REPORT-R2.md);
|
figures are pessimistic. Full trip report in [`../loadtest/REPORT.md`](../loadtest/REPORT.md);
|
||||||
it feeds R3 (h2c `MaxConcurrentStreams`/timeouts, body-size cap), R6 and R7 (per-player transports,
|
it feeds R3 (h2c `MaxConcurrentStreams`/timeouts, body-size cap), R6 and R7 (per-player transports,
|
||||||
separate hardware, pool/limit sizing).
|
separate hardware, pool/limit sizing).
|
||||||
- **CI:** `./loadtest/...` added to the path filter + vet/build/test; `go.work.sum` carries the new deps.
|
- **CI:** `./loadtest/...` added to the path filter + vet/build/test; `go.work.sum` carries the new deps.
|
||||||
@@ -454,7 +454,12 @@ Then Stage 18.
|
|||||||
one connection per player it bursts into its 2-core cap (the residual 2.49 % `transport_error`); backend
|
one connection per player it bursts into its 2-core cap (the residual 2.49 % `transport_error`); backend
|
||||||
~0.85 core and postgres ~1.4 cores had headroom; **tempo reached its 1 GiB cap**; the backend pool sat at
|
~0.85 core and postgres ~1.4 cores had headroom; **tempo reached its 1 GiB cap**; the backend pool sat at
|
||||||
its `MaxOpenConns=25` cap (28 backends); docker logs were unbounded (~14 MiB / 30 min on the backend at
|
its `MaxOpenConns=25` cap (28 backends); docker logs were unbounded (~14 MiB / 30 min on the backend at
|
||||||
info). Full write-up in [`../loadtest/REPORT-R7.md`](../loadtest/REPORT-R7.md).
|
info). Full write-up in [`../loadtest/REPORT.md`](../loadtest/REPORT.md). *(Superseded in part: a
|
||||||
|
later pass modelling the `game.evaluate` hot path traced the gateway's CPU appetite to
|
||||||
|
**gateway→backend connection churn** — the default 2-idle-connection HTTP transport — not proxying
|
||||||
|
work. Pooling the connections cut peak gateway CPU ~7× (~1.75 → ~0.26 cores at 500 players) and
|
||||||
|
removed the ephemeral-port-exhaustion cliff behind the residual `transport_error`, so the gateway is
|
||||||
|
no longer the binding constraint — postgres is. The 3-core gateway cap below is now generous headroom.)*
|
||||||
- **Round-2 tuning (owner-agreed, all in `deploy/docker-compose.yml`, no code change):** gateway **2 → 3
|
- **Round-2 tuning (owner-agreed, all in `deploy/docker-compose.yml`, no code change):** gateway **2 → 3
|
||||||
cores + `GOMAXPROCS=3`**; tempo memory **1 → 2 GiB**; backend `MAX_OPEN_CONNS` **25 → 40**; a json-file
|
cores + `GOMAXPROCS=3`**; tempo memory **1 → 2 GiB**; backend `MAX_OPEN_CONNS` **25 → 40**; a json-file
|
||||||
**log-rotation** default (10m × 3) applied contour-wide via a YAML anchor (level stays info).
|
**log-rotation** default (10m × 3) applied contour-wide via a YAML anchor (level stays info).
|
||||||
@@ -464,7 +469,7 @@ Then Stage 18.
|
|||||||
**burst** run (a single 100 → 500 jump) pegged the gateway at 3 cores (≈296 % sustained, 9.27 % error),
|
**burst** run (a single 100 → 500 jump) pegged the gateway at 3 cores (≈296 % sustained, 9.27 % error),
|
||||||
confirming it is **connection-CPU-bound** — a true arrival spike is a **horizontal-scaling** lever, not
|
confirming it is **connection-CPU-bound** — a true arrival spike is a **horizontal-scaling** lever, not
|
||||||
more cores per node (recorded in the prod-sizing recommendation).
|
more cores per node (recorded in the prod-sizing recommendation).
|
||||||
- **No schema change → no contour DB wipe.** Bake-back: `loadtest/REPORT-R7.md` (new), `loadtest/README.md`,
|
- **No schema change → no contour DB wipe.** Bake-back: `loadtest/REPORT.md`, `loadtest/README.md`,
|
||||||
`docs/TESTING.md`, the telemetry/observability section of `docs/ARCHITECTURE.md`, the repo-layout line in `CLAUDE.md`.
|
`docs/TESTING.md`, the telemetry/observability section of `docs/ARCHITECTURE.md`, the repo-layout line in `CLAUDE.md`.
|
||||||
|
|
||||||
- **UI — Tab-bar navigation redesign** (owner ad-hoc, not on the raw TODO list): drop the hamburger
|
- **UI — Tab-bar navigation redesign** (owner ad-hoc, not on the raw TODO list): drop the hamburger
|
||||||
|
|||||||
+3
-1
@@ -33,7 +33,9 @@ COPY backend ./backend
|
|||||||
# Reduce the workspace to what the backend needs: backend + pkg. loadtest and the
|
# Reduce the workspace to what the backend needs: backend + pkg. loadtest and the
|
||||||
# gateway replace it requires are not in this context, so drop both.
|
# gateway replace it requires are not in this context, so drop both.
|
||||||
RUN go work edit -dropuse=./gateway -dropuse=./platform/telegram -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0
|
RUN go work edit -dropuse=./gateway -dropuse=./platform/telegram -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0
|
||||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/backend ./backend/cmd/backend
|
# VERSION (the deploy passes the git tag) is stamped into the binary via the linker.
|
||||||
|
ARG VERSION=dev
|
||||||
|
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/backend ./backend/cmd/backend
|
||||||
|
|
||||||
# --- runtime -----------------------------------------------------------------
|
# --- runtime -----------------------------------------------------------------
|
||||||
FROM gcr.io/distroless/static-debian12:nonroot
|
FROM gcr.io/distroless/static-debian12:nonroot
|
||||||
|
|||||||
@@ -63,6 +63,7 @@ type gameCache struct {
|
|||||||
|
|
||||||
type cachedGame struct {
|
type cachedGame struct {
|
||||||
game *engine.Game
|
game *engine.Game
|
||||||
|
seats []Seat
|
||||||
variant string
|
variant string
|
||||||
lastAccess time.Time
|
lastAccess time.Time
|
||||||
}
|
}
|
||||||
@@ -71,24 +72,27 @@ func newGameCache(ttl time.Duration, now func() time.Time) *gameCache {
|
|||||||
return &gameCache{entries: make(map[uuid.UUID]*cachedGame), ttl: ttl, now: now}
|
return &gameCache{entries: make(map[uuid.UUID]*cachedGame), ttl: ttl, now: now}
|
||||||
}
|
}
|
||||||
|
|
||||||
// get returns the live game for id and refreshes its idle timer, or (nil, false).
|
// get returns the live game and its immutable seat list for id and refreshes its idle
|
||||||
func (c *gameCache) get(id uuid.UUID) (*engine.Game, bool) {
|
// timer, or (nil, nil, false). The seats let a read check membership (and label seats)
|
||||||
|
// without re-loading the game from the store, since seats never change after a game starts.
|
||||||
|
func (c *gameCache) get(id uuid.UUID) (*engine.Game, []Seat, bool) {
|
||||||
c.mu.Lock()
|
c.mu.Lock()
|
||||||
defer c.mu.Unlock()
|
defer c.mu.Unlock()
|
||||||
e, ok := c.entries[id]
|
e, ok := c.entries[id]
|
||||||
if !ok {
|
if !ok {
|
||||||
return nil, false
|
return nil, nil, false
|
||||||
}
|
}
|
||||||
e.lastAccess = c.now()
|
e.lastAccess = c.now()
|
||||||
return e.game, true
|
return e.game, e.seats, true
|
||||||
}
|
}
|
||||||
|
|
||||||
// put stores g as the live game for id. variant labels the entry so the active-
|
// put stores g as the live game for id together with its seat list. variant labels the
|
||||||
// games gauge can report counts by variant without inspecting engine internals.
|
// entry so the active-games gauge can report counts by variant without inspecting engine
|
||||||
func (c *gameCache) put(id uuid.UUID, g *engine.Game, variant string) {
|
// internals; seats are the game's immutable seat standings for the membership fast path.
|
||||||
|
func (c *gameCache) put(id uuid.UUID, g *engine.Game, variant string, seats []Seat) {
|
||||||
c.mu.Lock()
|
c.mu.Lock()
|
||||||
defer c.mu.Unlock()
|
defer c.mu.Unlock()
|
||||||
c.entries[id] = &cachedGame{game: g, variant: variant, lastAccess: c.now()}
|
c.entries[id] = &cachedGame{game: g, seats: seats, variant: variant, lastAccess: c.now()}
|
||||||
}
|
}
|
||||||
|
|
||||||
// remove drops id from the cache (used on a finished game and after a failed
|
// remove drops id from the cache (used on a finished game and after a failed
|
||||||
|
|||||||
@@ -94,8 +94,8 @@ func TestGameCacheEviction(t *testing.T) {
|
|||||||
cur := time.Unix(1_700_000_000, 0)
|
cur := time.Unix(1_700_000_000, 0)
|
||||||
cache := newGameCache(time.Hour, func() time.Time { return cur })
|
cache := newGameCache(time.Hour, func() time.Time { return cur })
|
||||||
id := uuid.New()
|
id := uuid.New()
|
||||||
cache.put(id, nil, "scrabble_en")
|
cache.put(id, nil, "scrabble_en", nil)
|
||||||
if _, ok := cache.get(id); !ok {
|
if _, _, ok := cache.get(id); !ok {
|
||||||
t.Fatal("game must be resident after put")
|
t.Fatal("game must be resident after put")
|
||||||
}
|
}
|
||||||
cur = cur.Add(30 * time.Minute)
|
cur = cur.Add(30 * time.Minute)
|
||||||
@@ -104,7 +104,7 @@ func TestGameCacheEviction(t *testing.T) {
|
|||||||
if n := cache.sweep(); n != 1 {
|
if n := cache.sweep(); n != 1 {
|
||||||
t.Errorf("sweep evicted %d, want 1", n)
|
t.Errorf("sweep evicted %d, want 1", n)
|
||||||
}
|
}
|
||||||
if _, ok := cache.get(id); ok {
|
if _, _, ok := cache.get(id); ok {
|
||||||
t.Error("game must be evicted after idle TTL")
|
t.Error("game must be evicted after idle TTL")
|
||||||
}
|
}
|
||||||
if cache.size() != 0 {
|
if cache.size() != 0 {
|
||||||
|
|||||||
@@ -287,12 +287,12 @@ func (svc *Service) Create(ctx context.Context, params CreateParams) (Game, erro
|
|||||||
if err := svc.store.CreateGame(ctx, ins, seats, seeding.draws); err != nil {
|
if err := svc.store.CreateGame(ctx, ins, seats, seeding.draws); err != nil {
|
||||||
return Game{}, err
|
return Game{}, err
|
||||||
}
|
}
|
||||||
svc.cache.put(id, g, params.Variant.String())
|
|
||||||
svc.metrics.recordStarted(ctx, params.Variant, params.VsAI)
|
svc.metrics.recordStarted(ctx, params.Variant, params.VsAI)
|
||||||
created, err := svc.store.GetGame(ctx, id)
|
created, err := svc.store.GetGame(ctx, id)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return Game{}, err
|
return Game{}, err
|
||||||
}
|
}
|
||||||
|
svc.cache.put(id, g, params.Variant.String(), created.Seats)
|
||||||
// Honest-AI game seated with a robot: if the robot moves first, reply at once
|
// Honest-AI game seated with a robot: if the robot moves first, reply at once
|
||||||
// (the periodic driver is the fallback). No-op for every human-only game.
|
// (the periodic driver is the fallback). No-op for every human-only game.
|
||||||
svc.triggerAI(created)
|
svc.triggerAI(created)
|
||||||
@@ -890,26 +890,35 @@ func (svc *Service) timeoutGame(ctx context.Context, gameID uuid.UUID, now time.
|
|||||||
// EvaluatePlay previews a tentative play for a seated player against the current
|
// EvaluatePlay previews a tentative play for a seated player against the current
|
||||||
// board without committing it: whether it is legal and what it would score.
|
// board without committing it: whether it is legal and what it would score.
|
||||||
func (svc *Service) EvaluatePlay(ctx context.Context, gameID, accountID uuid.UUID, tiles []engine.TileRecord) (EvalResult, error) {
|
func (svc *Service) EvaluatePlay(ctx context.Context, gameID, accountID uuid.UUID, tiles []engine.TileRecord) (EvalResult, error) {
|
||||||
|
unlock := svc.locks.lock(gameID)
|
||||||
|
defer unlock()
|
||||||
|
|
||||||
|
// Hot path: an active game stays cached — the engine game is mutated in place across
|
||||||
|
// moves and evicted only when it finishes — so on a hit the cached live game and its
|
||||||
|
// immutable seat list answer the membership check and the score with no DB read. This
|
||||||
|
// preview is fired on every tile placement, the hottest gameplay call at scale.
|
||||||
|
g, seats, ok := svc.cache.get(gameID)
|
||||||
|
if !ok {
|
||||||
|
// Cold path: load and validate from the store, then replay into the cache.
|
||||||
pre, err := svc.store.GetGame(ctx, gameID)
|
pre, err := svc.store.GetGame(ctx, gameID)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return EvalResult{}, err
|
return EvalResult{}, err
|
||||||
}
|
}
|
||||||
if _, ok := pre.seatOf(accountID); !ok {
|
|
||||||
return EvalResult{}, ErrNotAPlayer
|
|
||||||
}
|
|
||||||
if pre.Status == StatusFinished {
|
if pre.Status == StatusFinished {
|
||||||
return EvalResult{}, ErrFinished
|
return EvalResult{}, ErrFinished
|
||||||
}
|
}
|
||||||
|
if g, err = svc.liveGame(ctx, pre); err != nil {
|
||||||
unlock := svc.locks.lock(gameID)
|
|
||||||
defer unlock()
|
|
||||||
g, err := svc.liveGame(ctx, pre)
|
|
||||||
if err != nil {
|
|
||||||
return EvalResult{}, err
|
return EvalResult{}, err
|
||||||
}
|
}
|
||||||
|
seats = pre.Seats
|
||||||
|
}
|
||||||
|
if !seatedIn(seats, accountID) {
|
||||||
|
return EvalResult{}, ErrNotAPlayer
|
||||||
|
}
|
||||||
|
|
||||||
validateStart := time.Now()
|
validateStart := time.Now()
|
||||||
rec, err := g.EvaluatePlay(tiles)
|
rec, err := g.EvaluatePlay(tiles)
|
||||||
svc.metrics.recordValidate(ctx, pre.Variant, validateStart)
|
svc.metrics.recordValidate(ctx, g.Variant(), validateStart)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
if errors.Is(err, engine.ErrIllegalPlay) {
|
if errors.Is(err, engine.ErrIllegalPlay) {
|
||||||
return EvalResult{Valid: false}, nil
|
return EvalResult{Valid: false}, nil
|
||||||
@@ -1359,7 +1368,7 @@ func (svc *Service) ExportGCG(ctx context.Context, gameID uuid.UUID) (string, er
|
|||||||
// liveGame returns the live engine.Game for pre, rebuilding it from the journal
|
// liveGame returns the live engine.Game for pre, rebuilding it from the journal
|
||||||
// on a cache miss. Callers must hold the per-game lock.
|
// on a cache miss. Callers must hold the per-game lock.
|
||||||
func (svc *Service) liveGame(ctx context.Context, pre Game) (*engine.Game, error) {
|
func (svc *Service) liveGame(ctx context.Context, pre Game) (*engine.Game, error) {
|
||||||
if g, ok := svc.cache.get(pre.ID); ok {
|
if g, _, ok := svc.cache.get(pre.ID); ok {
|
||||||
return g, nil
|
return g, nil
|
||||||
}
|
}
|
||||||
g, err := svc.replay(ctx, pre)
|
g, err := svc.replay(ctx, pre)
|
||||||
@@ -1374,7 +1383,7 @@ func (svc *Service) liveGame(ctx context.Context, pre Game) (*engine.Game, error
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
if !g.Over() {
|
if !g.Over() {
|
||||||
svc.cache.put(pre.ID, g, pre.Variant.String())
|
svc.cache.put(pre.ID, g, pre.Variant.String(), pre.Seats)
|
||||||
}
|
}
|
||||||
return g, nil
|
return g, nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -355,27 +355,33 @@ func (s *Store) ExpiredOpen(ctx context.Context, now time.Time) ([]OpenGame, err
|
|||||||
// GetGame loads the games row joined with its seats (ordered by seat), or
|
// GetGame loads the games row joined with its seats (ordered by seat), or
|
||||||
// ErrNotFound.
|
// ErrNotFound.
|
||||||
func (s *Store) GetGame(ctx context.Context, id uuid.UUID) (Game, error) {
|
func (s *Store) GetGame(ctx context.Context, id uuid.UUID) (Game, error) {
|
||||||
gstmt := postgres.SELECT(table.Games.AllColumns).
|
// One round-trip: the game joined with its seats. A LEFT JOIN keeps a (would-be)
|
||||||
FROM(table.Games).
|
// seatless game returning the game with no seats, exactly as the prior two-query
|
||||||
|
// version did; ORDER BY seat preserves seat order. The games columns repeat per seat
|
||||||
|
// row — cheap at 2-4 seats, and one round-trip instead of two, which matters because
|
||||||
|
// GetGame is the universal "load the game" step on every game operation.
|
||||||
|
stmt := postgres.SELECT(table.Games.AllColumns, table.GamePlayers.AllColumns).
|
||||||
|
FROM(table.Games.LEFT_JOIN(table.GamePlayers, table.GamePlayers.GameID.EQ(table.Games.GameID))).
|
||||||
WHERE(table.Games.GameID.EQ(postgres.UUID(id))).
|
WHERE(table.Games.GameID.EQ(postgres.UUID(id))).
|
||||||
LIMIT(1)
|
ORDER_BY(table.GamePlayers.Seat.ASC())
|
||||||
var grow model.Games
|
var rows []struct {
|
||||||
if err := gstmt.QueryContext(ctx, s.db, &grow); err != nil {
|
model.Games
|
||||||
if errors.Is(err, qrm.ErrNoRows) {
|
model.GamePlayers
|
||||||
return Game{}, ErrNotFound
|
|
||||||
}
|
}
|
||||||
|
if err := stmt.QueryContext(ctx, s.db, &rows); err != nil {
|
||||||
return Game{}, fmt.Errorf("game: get %s: %w", id, err)
|
return Game{}, fmt.Errorf("game: get %s: %w", id, err)
|
||||||
}
|
}
|
||||||
|
if len(rows) == 0 {
|
||||||
sstmt := postgres.SELECT(table.GamePlayers.AllColumns).
|
return Game{}, ErrNotFound
|
||||||
FROM(table.GamePlayers).
|
|
||||||
WHERE(table.GamePlayers.GameID.EQ(postgres.UUID(id))).
|
|
||||||
ORDER_BY(table.GamePlayers.Seat.ASC())
|
|
||||||
var srows []model.GamePlayers
|
|
||||||
if err := sstmt.QueryContext(ctx, s.db, &srows); err != nil {
|
|
||||||
return Game{}, fmt.Errorf("game: get seats %s: %w", id, err)
|
|
||||||
}
|
}
|
||||||
return projectGame(grow, srows)
|
seats := make([]model.GamePlayers, 0, len(rows))
|
||||||
|
for i := range rows {
|
||||||
|
// Skip the phantom all-NULL seat row a LEFT JOIN yields for a seatless game.
|
||||||
|
if rows[i].GamePlayers.GameID == id {
|
||||||
|
seats = append(seats, rows[i].GamePlayers)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return projectGame(rows[0].Games, seats)
|
||||||
}
|
}
|
||||||
|
|
||||||
// GetGameVariant reads just a game's variant — a cheap single-column lookup the edge uses
|
// GetGameVariant reads just a game's variant — a cheap single-column lookup the edge uses
|
||||||
|
|||||||
@@ -184,6 +184,18 @@ func (g Game) seatOf(accountID uuid.UUID) (int, bool) {
|
|||||||
return 0, false
|
return 0, false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// seatedIn reports whether accountID holds a seat in seats. It backs the read-side
|
||||||
|
// membership check against the cached, immutable seat list, so a hot read can skip
|
||||||
|
// loading the game from the store.
|
||||||
|
func seatedIn(seats []Seat, accountID uuid.UUID) bool {
|
||||||
|
for _, s := range seats {
|
||||||
|
if s.AccountID == accountID {
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
return false
|
||||||
|
}
|
||||||
|
|
||||||
// MoveResult is the outcome of a committed transition: the decoded move and the
|
// MoveResult is the outcome of a committed transition: the decoded move and the
|
||||||
// post-move game, plus the actor's own refilled rack and the bag size after the draw
|
// post-move game, plus the actor's own refilled rack and the bag size after the draw
|
||||||
// (Rack/BagLen), so the mover renders the next state from the response without a
|
// (Rack/BagLen), so the mover renders the next state from the response without a
|
||||||
|
|||||||
@@ -543,6 +543,12 @@ func TestEvaluatePlayPreview(t *testing.T) {
|
|||||||
if bad.Valid {
|
if bad.Valid {
|
||||||
t.Error("disconnected play must be invalid")
|
t.Error("disconnected play must be invalid")
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// A non-seated account cannot preview: with the game warm in the live cache, the
|
||||||
|
// membership check runs against the cached seat list (the hot path that skips GetGame).
|
||||||
|
if _, err := svc.EvaluatePlay(ctx, g.ID, provisionAccount(t), hint.Tiles); !errors.Is(err, game.ErrNotAPlayer) {
|
||||||
|
t.Errorf("evaluate by a non-player = %v, want ErrNotAPlayer", err)
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// TestConcurrentSubmitSerialized confirms the per-game lock lets only one of two
|
// TestConcurrentSubmitSerialized confirms the per-game lock lets only one of two
|
||||||
|
|||||||
+69
-3
@@ -17,11 +17,12 @@ operational reference for **every environment variable**.
|
|||||||
| `backend` | built (`backend/Dockerfile`) | Domain service; bakes in the DAWG dictionaries; runs migrations at boot. |
|
| `backend` | built (`backend/Dockerfile`) | Domain service; bakes in the DAWG dictionaries; runs migrations at boot. |
|
||||||
| `postgres` | `postgres:17-alpine` | Database (named volume, `pg_isready` healthcheck). |
|
| `postgres` | `postgres:17-alpine` | Database (named volume, `pg_isready` healthcheck). |
|
||||||
| `validator` | built (`platform/telegram/Dockerfile`, target `validator`) | Telegram HMAC validator (no VPN, no Bot API); internal gRPC at `validator:9091`. Game login depends only on this. |
|
| `validator` | built (`platform/telegram/Dockerfile`, target `validator`) | Telegram HMAC validator (no VPN, no Bot API); internal gRPC at `validator:9091`. Game login depends only on this. |
|
||||||
| `vpn` + `bot` | sidecar + built (`platform/telegram/Dockerfile`, target `bot`) | Telegram bot; egresses through the AmneziaWG sidecar; holds no inbound port — dials the gateway bot-link (mTLS) at `gateway:9443`. |
|
| `vpn` + `bot` | sidecar + built (`platform/telegram/Dockerfile`, target `bot`) | Telegram bot, gated to the **`telegram-local`** profile; egresses through the AmneziaWG sidecar and dials the gateway bot-link (mTLS) at `gateway:9443`. The test contour activates the profile; the prod **main** host omits it and runs the bot standalone on its **own host** (`docker-compose.bot.yml`, no VPN — native Bot API egress). |
|
||||||
| `otelcol` | `otel/opentelemetry-collector-contrib` | OTLP/gRPC `:4317` → Prometheus scrape (`:9464`) + Tempo. |
|
| `otelcol` | `otel/opentelemetry-collector-contrib` | OTLP/gRPC `:4317` → Prometheus scrape (`:9464`) + Tempo. |
|
||||||
| `prometheus` | `prom/prometheus` | Metrics, 15d retention. |
|
| `prometheus` | `prom/prometheus` | Metrics, 15d retention (7d in prod). |
|
||||||
| `tempo` | `grafana/tempo` | Traces, 72h retention. |
|
| `tempo` | `grafana/tempo` | Traces, 72h retention. |
|
||||||
| `grafana` | `grafana/grafana` | Dashboards (provisioned), anonymous-admin behind caddy's `/_gm/grafana`. |
|
| `grafana` | `grafana/grafana` | Dashboards (provisioned), anonymous-admin behind caddy's `/_gm/grafana`. |
|
||||||
|
| `node_exporter` | `quay.io/prometheus/node-exporter` | Host CPU/memory/disk metrics (Prometheus job `node`); the OOM signal on the tight prod main host (2 vCPU / 1.9 GiB). |
|
||||||
|
|
||||||
Networking: inter-service traffic is on the private `internal` network
|
Networking: inter-service traffic is on the private `internal` network
|
||||||
(project-scoped DNS); only `caddy` joins the shared external `edge` network so the
|
(project-scoped DNS); only `caddy` joins the shared external `edge` network so the
|
||||||
@@ -59,7 +60,6 @@ compose binds from this directory.
|
|||||||
| Variable | Gitea kind | Purpose |
|
| Variable | Gitea kind | Purpose |
|
||||||
| --- | --- | --- |
|
| --- | --- | --- |
|
||||||
| `POSTGRES_PASSWORD` | secret | Postgres password (also embedded in `BACKEND_POSTGRES_DSN`). |
|
| `POSTGRES_PASSWORD` | secret | Postgres password (also embedded in `BACKEND_POSTGRES_DSN`). |
|
||||||
| `AWG_CONF` | secret | AmneziaWG config for the VPN sidecar (the bot's only Telegram egress in the test contour). **Must not contain a `DNS=` line** — it hijacks the shared netns's resolv.conf and breaks the bot resolving `otelcol` / `gateway`. Without it, Docker's resolver handles `otelcol`, `gateway` and `api.telegram.org`. |
|
|
||||||
| `GM_BASICAUTH_HASH` | secret | bcrypt hash gating `/_gm` (admin console + Grafana). Generate with `docker run --rm caddy:2-alpine caddy hash-password --plaintext '<pw>'`. |
|
| `GM_BASICAUTH_HASH` | secret | bcrypt hash gating `/_gm` (admin console + Grafana). Generate with `docker run --rm caddy:2-alpine caddy hash-password --plaintext '<pw>'`. |
|
||||||
| `TELEGRAM_MINIAPP_URL` | variable | The Mini App URL the bot hands out in deep links / buttons. |
|
| `TELEGRAM_MINIAPP_URL` | variable | The Mini App URL the bot hands out in deep links / buttons. |
|
||||||
|
|
||||||
@@ -67,6 +67,13 @@ compose binds from this directory.
|
|||||||
secret) and the bot (Bot API). It defaults to empty in compose, but both **fail at
|
secret) and the bot (Bot API). It defaults to empty in compose, but both **fail at
|
||||||
boot** when it is empty.
|
boot** when it is empty.
|
||||||
|
|
||||||
|
**Conditionally — `AWG_CONF`** (secret): the AmneziaWG config for the VPN sidecar, needed
|
||||||
|
only when the `telegram-local` profile runs (the test contour and local runs with the
|
||||||
|
bot). It is **not** `:?`-guarded — compose interpolates profiled-out services too, so the
|
||||||
|
prod main host (no VPN) must not require it. It **must not contain a `DNS=` line** — that
|
||||||
|
hijacks the shared netns's resolv.conf and breaks the bot resolving `otelcol` / `gateway`;
|
||||||
|
without it Docker's resolver handles `otelcol`, `gateway` and `api.telegram.org`.
|
||||||
|
|
||||||
## Optional variables (with defaults)
|
## Optional variables (with defaults)
|
||||||
|
|
||||||
| Variable | Gitea kind | Default | Purpose |
|
| Variable | Gitea kind | Default | Purpose |
|
||||||
@@ -110,6 +117,65 @@ collector's / gateway's internal IP is fine (connected route), but its `AWG_CONF
|
|||||||
which resolves `otelcol`, `gateway` and `api.telegram.org`. `GATEWAY_ADMIN_*` is
|
which resolves `otelcol`, `gateway` and `api.telegram.org`. `GATEWAY_ADMIN_*` is
|
||||||
intentionally **unset** — caddy owns `/_gm` in the contour.
|
intentionally **unset** — caddy owns `/_gm` in the contour.
|
||||||
|
|
||||||
|
## Production rollout
|
||||||
|
|
||||||
|
Prod runs on **two hosts** (main = full stack + ACME on the domain; tg = the bot only,
|
||||||
|
native Bot API, no VPN), one-time provisioned by **[`ansible/`](ansible/)** (docker, a
|
||||||
|
non-sudo `deploy` user holding the CI key, key-only sshd, default-deny ufw, fail2ban).
|
||||||
|
Re-run `ansible/` after a host resize — it is idempotent.
|
||||||
|
|
||||||
|
**To roll out:** merge `development → master` (CI green), then run the **`prod-deploy`**
|
||||||
|
workflow manually (Gitea → Actions → prod-deploy → run from `master`, input
|
||||||
|
`confirm=deploy`). It builds + pushes the images to the registry, ships the
|
||||||
|
compose/config/certs/env over SSH, deploys the main host with `prod-deploy.sh` (rolling,
|
||||||
|
health-gated, **auto-rollback to the previous tag**), then the bot host, then probes the
|
||||||
|
public site. After `master` is green this workflow is the **only** thing that touches
|
||||||
|
prod — nothing auto-deploys there. It runs four visible jobs: **build → deploy-main →
|
||||||
|
deploy-bot → verify** (the per-service rolling shows in the deploy-main log).
|
||||||
|
|
||||||
|
**Versioning.** Each release is a git tag `vX.Y.Z` on `master`; the deploy stamps
|
||||||
|
`git describe --tags` into every image tag, every binary (`-ldflags` → `pkg/version` →
|
||||||
|
the `service.version` telemetry attribute) and the SPA About screen. Tag the release
|
||||||
|
before running the deploy:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
git tag -a v1.0.0 -m v1.0.0 && git push origin v1.0.0
|
||||||
|
```
|
||||||
|
|
||||||
|
**Manual rollback** (any time after a successful deploy). Run the **`prod-rollback`**
|
||||||
|
workflow (Gitea → Actions → prod-rollback, `confirm=rollback`). Leave `target_version`
|
||||||
|
blank to roll back to the previously deployed version (read from the host's
|
||||||
|
`PREVIOUS_TAG`), or set it to a release tag from the **Releases** page. It re-deploys
|
||||||
|
that already-published image rolling + health-gated — no rebuild, no DB migration
|
||||||
|
(image rollback is DB-safe under the expand-contract rule). The registry keeps every
|
||||||
|
release tag, so any prior release is reachable.
|
||||||
|
|
||||||
|
**Migrations** must be **expand-contract** (backward-compatible; goose is forward-only):
|
||||||
|
the automatic rollback is image-only and never restores the DB. A deploy that changes
|
||||||
|
`backend/internal/postgres/migrations/` opens a maintenance window — the backend (sole
|
||||||
|
writer) is stopped for a consistent `pg_dump` into `/opt/scrabble/dumps` before the new
|
||||||
|
backend migrates. **Manual DB restore** (only if a migration was destructive):
|
||||||
|
`docker exec -i scrabble-postgres psql -U scrabble -d scrabble -c 'DROP SCHEMA backend CASCADE'`,
|
||||||
|
then pipe the dump into the same `psql`, and redeploy the matching old tag.
|
||||||
|
|
||||||
|
**bot-link cert rotation:** regenerate (`deploy/gen-certs.sh /tmp/c --force`), reset the
|
||||||
|
five `PROD_BOTLINK_*` secrets from `/tmp/c`, and re-run the workflow — both hosts redeploy
|
||||||
|
together with the fresh CA.
|
||||||
|
|
||||||
|
**Sizing / monitoring:** the main host launches undersized (2 vCPU / 1.9 GiB); the prod
|
||||||
|
overlay trims limits + `GOMAXPROCS=2` + 7d Prometheus retention, and `node_exporter` feeds
|
||||||
|
host memory to Grafana (`/_gm/grafana/`). Watch host memory and resize at Selectel when
|
||||||
|
players arrive.
|
||||||
|
|
||||||
|
**`PROD_` Gitea set** (mirrors `TEST_`, mapped onto the unprefixed names above) — secrets:
|
||||||
|
`PROD_{POSTGRES_PASSWORD, GM_BASICAUTH_HASH, GRAFANA_ADMIN_PASSWORD, TELEGRAM_BOT_TOKEN,
|
||||||
|
TELEGRAM_PROMO_BOT_TOKEN, REGISTRY_PASSWORD, SSH_KEY, SSH_KNOWN_HOSTS, BOTLINK_CA,
|
||||||
|
BOTLINK_GATEWAY_CERT, BOTLINK_GATEWAY_KEY, BOTLINK_BOT_CERT, BOTLINK_BOT_KEY}`; variables:
|
||||||
|
`PROD_{REGISTRY_USER, MAIN_HOST, TG_HOST, CADDY_SITE_ADDRESS, GM_BASICAUTH_USER,
|
||||||
|
GRAFANA_ROOT_URL, LOG_LEVEL, DICT_VERSION, TELEGRAM_MINIAPP_URL, TELEGRAM_GAME_CHANNEL_ID,
|
||||||
|
TELEGRAM_CHAT_ID, TELEGRAM_BOT_USERNAME, VITE_TELEGRAM_BOT_ID, VITE_TELEGRAM_LINK,
|
||||||
|
VITE_TELEGRAM_GAME_CHANNEL_NAME}`.
|
||||||
|
|
||||||
## Host-side setup (outside this repo)
|
## Host-side setup (outside this repo)
|
||||||
|
|
||||||
- **`edge` network** must exist on the host (`docker network create edge`).
|
- **`edge` network** must exist on the host (`docker network create edge`).
|
||||||
|
|||||||
@@ -0,0 +1,49 @@
|
|||||||
|
# Prod host provisioning (Stage 18)
|
||||||
|
|
||||||
|
Idempotent Ansible that prepares the two production hosts. It installs Docker, a
|
||||||
|
non-sudo `deploy` service account, SSH hardening, a default-deny firewall,
|
||||||
|
fail2ban, unattended security upgrades and time sync. It does **not** deploy the
|
||||||
|
application — that is `.gitea/workflows/prod-deploy.yaml`'s job, running as the
|
||||||
|
`deploy` account this playbook creates.
|
||||||
|
|
||||||
|
Hosts are referenced by `~/.ssh/config` aliases (`scrabble-main-ops`,
|
||||||
|
`scrabble-tg-ops`), so no IPs or key paths live in the repo.
|
||||||
|
|
||||||
|
## Prerequisites (controller)
|
||||||
|
|
||||||
|
- `ansible` with the bundled collections (`community.general`, `community.docker`,
|
||||||
|
`ansible.posix`).
|
||||||
|
- The two hosts reachable as root via the ssh-config aliases, host keys already
|
||||||
|
accepted into `known_hosts` (`host_key_checking = True`).
|
||||||
|
|
||||||
|
## One-time: the CI deploy key
|
||||||
|
|
||||||
|
The CI prod-deploy workflow logs into the hosts as `deploy` using a dedicated
|
||||||
|
key. Generate it once on the controller, authorize its public half via the
|
||||||
|
playbook, and store its private half **only** in the Gitea `PROD_SSH_KEY` secret:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
ssh-keygen -t ed25519 -N '' -C scrabble-ci-deploy \
|
||||||
|
-f ~/.ssh/scrabble_ci_deploy_ed25519
|
||||||
|
# private half -> Gitea secret PROD_SSH_KEY (set via API); never commit it
|
||||||
|
```
|
||||||
|
|
||||||
|
## Run
|
||||||
|
|
||||||
|
```sh
|
||||||
|
cd deploy/ansible
|
||||||
|
ansible-playbook site.yml
|
||||||
|
```
|
||||||
|
|
||||||
|
The playbook reads the public key from `~/.ssh/scrabble_ci_deploy_ed25519.pub` by
|
||||||
|
default; override with `-e deploy_ci_pubkey_path=/path/to/key.pub`. Re-running is
|
||||||
|
safe (idempotent) and survives a host resize.
|
||||||
|
|
||||||
|
## What each host gets
|
||||||
|
|
||||||
|
- **both** (`common`): docker-ce + compose plugin, `daemon.json` (live-restore,
|
||||||
|
10m×3 log rotation), `deploy` user (docker group, no sudo), key-only sshd,
|
||||||
|
`ufw` default-deny incoming + allow SSH, fail2ban sshd jail, unattended
|
||||||
|
upgrades, chrony, `/opt/scrabble/{config,certs,dumps,images}`.
|
||||||
|
- **main**: `ufw` opens 80/443/9443; the external `edge` docker network.
|
||||||
|
- **tg**: verifies direct `api.telegram.org` egress (the no-VPN assumption).
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
[defaults]
|
||||||
|
inventory = inventory.ini
|
||||||
|
roles_path = roles
|
||||||
|
interpreter_python = /usr/bin/python3
|
||||||
|
host_key_checking = True
|
||||||
|
stdout_callback = yaml
|
||||||
|
deprecation_warnings = False
|
||||||
|
retry_files_enabled = False
|
||||||
|
|
||||||
|
[ssh_connection]
|
||||||
|
pipelining = True
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
---
|
||||||
|
# Service account the CI prod-deploy workflow uses to drive docker on the hosts.
|
||||||
|
# Membership in the docker group is root-equivalent (docker socket access), which
|
||||||
|
# is all the deploy workflow needs; the account is deliberately not given sudo.
|
||||||
|
deploy_user: deploy
|
||||||
|
|
||||||
|
# Public half of the dedicated CI deploy SSH key, read from the controller at run
|
||||||
|
# time. The private half is generated on the controller during provisioning and
|
||||||
|
# stored ONLY in the Gitea PROD_SSH_KEY secret; it is never committed. Override the
|
||||||
|
# path with -e deploy_ci_pubkey_path=/path/to/key.pub if the key lives elsewhere.
|
||||||
|
deploy_ci_pubkey_path: "{{ lookup('env', 'HOME') }}/.ssh/scrabble_ci_deploy_ed25519.pub"
|
||||||
|
deploy_ci_pubkey: "{{ lookup('file', deploy_ci_pubkey_path) }}"
|
||||||
|
|
||||||
|
# Base directory the deploy workflow rsyncs compose files, config, certs and dumps
|
||||||
|
# into. Owned by deploy_user so the workflow needs no elevation.
|
||||||
|
scrabble_base_dir: /opt/scrabble
|
||||||
|
|
||||||
|
# Docker daemon json-file log rotation, mirroring the compose x-logging anchor so
|
||||||
|
# the host's own containers (and any ad-hoc runs) rotate identically.
|
||||||
|
docker_log_max_size: "10m"
|
||||||
|
docker_log_max_file: "3"
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
# Production inventory for Stage 18.
|
||||||
|
#
|
||||||
|
# Hosts resolve through the operator's ~/.ssh/config aliases, so HostName (public
|
||||||
|
# IP), User and IdentityFile live there — no IPs or key paths are committed here.
|
||||||
|
# scrabble-main-ops -> main stack host (public IP, domain erudit-game.ru)
|
||||||
|
# scrabble-tg-ops -> Telegram bot host (direct Bot API egress, no VPN)
|
||||||
|
|
||||||
|
[main]
|
||||||
|
scrabble-main-ops
|
||||||
|
|
||||||
|
[tg]
|
||||||
|
scrabble-tg-ops
|
||||||
|
|
||||||
|
[prod:children]
|
||||||
|
main
|
||||||
|
tg
|
||||||
|
|
||||||
|
[prod:vars]
|
||||||
|
ansible_user=root
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
---
|
||||||
|
- name: restart docker
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: docker
|
||||||
|
state: restarted
|
||||||
|
|
||||||
|
- name: reload sshd
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: ssh
|
||||||
|
state: reloaded
|
||||||
|
|
||||||
|
- name: restart fail2ban
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: fail2ban
|
||||||
|
state: restarted
|
||||||
@@ -0,0 +1,167 @@
|
|||||||
|
---
|
||||||
|
# Common baseline applied to both prod hosts: Docker engine, a non-sudo deploy
|
||||||
|
# service account, SSH hardening, a default-deny firewall, fail2ban, unattended
|
||||||
|
# security upgrades and time sync. Every task is idempotent.
|
||||||
|
|
||||||
|
- name: Install base packages
|
||||||
|
ansible.builtin.apt:
|
||||||
|
name:
|
||||||
|
- ca-certificates
|
||||||
|
- curl
|
||||||
|
- gnupg
|
||||||
|
- ufw
|
||||||
|
- fail2ban
|
||||||
|
- unattended-upgrades
|
||||||
|
- chrony
|
||||||
|
state: present
|
||||||
|
update_cache: true
|
||||||
|
cache_valid_time: 3600
|
||||||
|
|
||||||
|
# --- Docker engine (official repo; trixie is published upstream) ---------------
|
||||||
|
|
||||||
|
- name: Create apt keyring directory
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: /etc/apt/keyrings
|
||||||
|
state: directory
|
||||||
|
mode: "0755"
|
||||||
|
|
||||||
|
- name: Install Docker apt GPG key
|
||||||
|
ansible.builtin.get_url:
|
||||||
|
url: https://download.docker.com/linux/debian/gpg
|
||||||
|
dest: /etc/apt/keyrings/docker.asc
|
||||||
|
mode: "0644"
|
||||||
|
|
||||||
|
- name: Add Docker apt repository
|
||||||
|
ansible.builtin.apt_repository:
|
||||||
|
repo: >-
|
||||||
|
deb [arch=amd64 signed-by=/etc/apt/keyrings/docker.asc]
|
||||||
|
https://download.docker.com/linux/debian {{ ansible_distribution_release }} stable
|
||||||
|
filename: docker
|
||||||
|
state: present
|
||||||
|
|
||||||
|
- name: Install Docker engine and the compose plugin
|
||||||
|
ansible.builtin.apt:
|
||||||
|
name:
|
||||||
|
- docker-ce
|
||||||
|
- docker-ce-cli
|
||||||
|
- containerd.io
|
||||||
|
- docker-buildx-plugin
|
||||||
|
- docker-compose-plugin
|
||||||
|
state: present
|
||||||
|
update_cache: true
|
||||||
|
|
||||||
|
- name: Configure the Docker daemon (live-restore + log rotation)
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: daemon.json.j2
|
||||||
|
dest: /etc/docker/daemon.json
|
||||||
|
mode: "0644"
|
||||||
|
notify: restart docker
|
||||||
|
|
||||||
|
- name: Enable and start Docker
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: docker
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
|
||||||
|
# --- Deploy service account ----------------------------------------------------
|
||||||
|
|
||||||
|
- name: Create the deploy service account
|
||||||
|
ansible.builtin.user:
|
||||||
|
name: "{{ deploy_user }}"
|
||||||
|
groups: docker
|
||||||
|
append: true
|
||||||
|
shell: /bin/bash
|
||||||
|
create_home: true
|
||||||
|
|
||||||
|
- name: Ensure the deploy .ssh directory
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "/home/{{ deploy_user }}/.ssh"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ deploy_user }}"
|
||||||
|
group: "{{ deploy_user }}"
|
||||||
|
mode: "0700"
|
||||||
|
|
||||||
|
- name: Authorize the CI deploy SSH key (exclusive)
|
||||||
|
ansible.builtin.copy:
|
||||||
|
dest: "/home/{{ deploy_user }}/.ssh/authorized_keys"
|
||||||
|
content: "{{ deploy_ci_pubkey }}\n"
|
||||||
|
owner: "{{ deploy_user }}"
|
||||||
|
group: "{{ deploy_user }}"
|
||||||
|
mode: "0600"
|
||||||
|
|
||||||
|
# --- SSH hardening -------------------------------------------------------------
|
||||||
|
|
||||||
|
- name: Harden sshd (key-only auth)
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: sshd-hardening.conf.j2
|
||||||
|
dest: /etc/ssh/sshd_config.d/10-scrabble-hardening.conf
|
||||||
|
mode: "0644"
|
||||||
|
validate: sshd -t -f %s
|
||||||
|
notify: reload sshd
|
||||||
|
|
||||||
|
# --- Firewall (default deny incoming) ------------------------------------------
|
||||||
|
# SSH is allowed before the policy flips so enabling ufw never locks us out.
|
||||||
|
|
||||||
|
- name: Allow SSH through the firewall
|
||||||
|
community.general.ufw:
|
||||||
|
rule: allow
|
||||||
|
name: OpenSSH
|
||||||
|
|
||||||
|
- name: Default-deny incoming, allow outgoing
|
||||||
|
community.general.ufw:
|
||||||
|
direction: "{{ item.direction }}"
|
||||||
|
policy: "{{ item.policy }}"
|
||||||
|
loop:
|
||||||
|
- { direction: incoming, policy: deny }
|
||||||
|
- { direction: outgoing, policy: allow }
|
||||||
|
|
||||||
|
- name: Enable the firewall
|
||||||
|
community.general.ufw:
|
||||||
|
state: enabled
|
||||||
|
|
||||||
|
# --- fail2ban ------------------------------------------------------------------
|
||||||
|
|
||||||
|
- name: Configure the fail2ban sshd jail
|
||||||
|
ansible.builtin.template:
|
||||||
|
src: jail.local.j2
|
||||||
|
dest: /etc/fail2ban/jail.local
|
||||||
|
mode: "0644"
|
||||||
|
notify: restart fail2ban
|
||||||
|
|
||||||
|
- name: Enable and start fail2ban
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: fail2ban
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
|
||||||
|
# --- Unattended security upgrades + time sync ----------------------------------
|
||||||
|
|
||||||
|
- name: Enable unattended upgrades
|
||||||
|
ansible.builtin.copy:
|
||||||
|
dest: /etc/apt/apt.conf.d/20auto-upgrades
|
||||||
|
mode: "0644"
|
||||||
|
content: |
|
||||||
|
APT::Periodic::Update-Package-Lists "1";
|
||||||
|
APT::Periodic::Unattended-Upgrade "1";
|
||||||
|
|
||||||
|
- name: Enable and start chrony
|
||||||
|
ansible.builtin.service:
|
||||||
|
name: chrony
|
||||||
|
enabled: true
|
||||||
|
state: started
|
||||||
|
|
||||||
|
# --- Deploy directories --------------------------------------------------------
|
||||||
|
|
||||||
|
- name: Create the scrabble base directories
|
||||||
|
ansible.builtin.file:
|
||||||
|
path: "{{ scrabble_base_dir }}/{{ item }}"
|
||||||
|
state: directory
|
||||||
|
owner: "{{ deploy_user }}"
|
||||||
|
group: "{{ deploy_user }}"
|
||||||
|
mode: "0750"
|
||||||
|
loop:
|
||||||
|
- ""
|
||||||
|
- config
|
||||||
|
- certs
|
||||||
|
- dumps
|
||||||
|
- images
|
||||||
@@ -0,0 +1,8 @@
|
|||||||
|
{
|
||||||
|
"live-restore": true,
|
||||||
|
"log-driver": "json-file",
|
||||||
|
"log-opts": {
|
||||||
|
"max-size": "{{ docker_log_max_size }}",
|
||||||
|
"max-file": "{{ docker_log_max_file }}"
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,9 @@
|
|||||||
|
# Managed by Ansible (deploy/ansible).
|
||||||
|
[DEFAULT]
|
||||||
|
bantime = 1h
|
||||||
|
findtime = 10m
|
||||||
|
maxretry = 5
|
||||||
|
backend = systemd
|
||||||
|
|
||||||
|
[sshd]
|
||||||
|
enabled = true
|
||||||
@@ -0,0 +1,6 @@
|
|||||||
|
# Managed by Ansible (deploy/ansible). Key-only authentication.
|
||||||
|
# root stays reachable by key (prohibit-password) for provisioning re-runs.
|
||||||
|
PasswordAuthentication no
|
||||||
|
PermitRootLogin prohibit-password
|
||||||
|
PubkeyAuthentication yes
|
||||||
|
KbdInteractiveAuthentication no
|
||||||
@@ -0,0 +1,18 @@
|
|||||||
|
---
|
||||||
|
# Main stack host: public web + bot-link ports and the external 'edge' network
|
||||||
|
# the compose stack attaches caddy to.
|
||||||
|
|
||||||
|
- name: Open public web and bot-link ports
|
||||||
|
community.general.ufw:
|
||||||
|
rule: allow
|
||||||
|
port: "{{ item }}"
|
||||||
|
proto: tcp
|
||||||
|
loop:
|
||||||
|
- "80" # HTTP (ACME challenge + redirect to HTTPS)
|
||||||
|
- "443" # HTTPS (caddy edge)
|
||||||
|
- "9443" # bot-link mTLS (remote bot dials in; mutual TLS gates access)
|
||||||
|
|
||||||
|
- name: Ensure the external 'edge' docker network exists
|
||||||
|
community.docker.docker_network:
|
||||||
|
name: edge
|
||||||
|
state: present
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
---
|
||||||
|
# Telegram bot host: holds no inbound port beyond SSH (the bot dials out to the
|
||||||
|
# Bot API and into the main host's bot-link). We only verify direct Bot API
|
||||||
|
# egress here, since the "no VPN" decision depends on it.
|
||||||
|
|
||||||
|
- name: Verify direct Telegram Bot API egress (no VPN on this host)
|
||||||
|
ansible.builtin.uri:
|
||||||
|
url: https://api.telegram.org/
|
||||||
|
method: GET
|
||||||
|
status_code: [200, 301, 302, 401, 404] # any HTTP reply proves reachability
|
||||||
|
timeout: 10
|
||||||
|
register: tg_egress
|
||||||
|
failed_when: false
|
||||||
|
|
||||||
|
- name: Report Telegram reachability
|
||||||
|
ansible.builtin.debug:
|
||||||
|
msg: >-
|
||||||
|
api.telegram.org reachable:
|
||||||
|
{{ (tg_egress.status | default(0) | int) > 0 }} (status {{ tg_egress.status | default('none') }})
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
---
|
||||||
|
# Stage 18 host provisioning. Idempotent: safe to re-run after a host resize.
|
||||||
|
# Prepares hosts only (docker, hardening, service account, firewall); the
|
||||||
|
# application is deployed separately by .gitea/workflows/prod-deploy.yaml.
|
||||||
|
|
||||||
|
- name: Common baseline (both hosts)
|
||||||
|
hosts: prod
|
||||||
|
become: true
|
||||||
|
pre_tasks:
|
||||||
|
- name: Require a well-formed CI deploy public key
|
||||||
|
ansible.builtin.assert:
|
||||||
|
that:
|
||||||
|
- deploy_ci_pubkey | length > 0
|
||||||
|
- deploy_ci_pubkey is search('^(ssh|ecdsa)-')
|
||||||
|
fail_msg: >-
|
||||||
|
deploy_ci_pubkey is empty or malformed. Generate the key first
|
||||||
|
(see deploy/ansible/README.md) or override deploy_ci_pubkey_path.
|
||||||
|
roles:
|
||||||
|
- common
|
||||||
|
|
||||||
|
- name: Main stack host
|
||||||
|
hosts: main
|
||||||
|
become: true
|
||||||
|
roles:
|
||||||
|
- main
|
||||||
|
|
||||||
|
- name: Telegram bot host
|
||||||
|
hosts: tg
|
||||||
|
become: true
|
||||||
|
roles:
|
||||||
|
- tg
|
||||||
@@ -0,0 +1,53 @@
|
|||||||
|
# Production Telegram bot host descriptor (standalone — NOT an overlay). Run only on
|
||||||
|
# the bot host:
|
||||||
|
# docker compose -f docker-compose.bot.yml up -d
|
||||||
|
#
|
||||||
|
# The bot egresses to the Bot API directly (no VPN sidecar) and dials the main host's
|
||||||
|
# published bot-link :9443 over mTLS. It exports no telemetry — otelcol lives on the
|
||||||
|
# main host and is unreachable from here — so observe it via `docker logs` on this host.
|
||||||
|
# Values come from the prod-deploy workflow (PROD_ secrets/variables); BOT_IMAGE is the
|
||||||
|
# pushed registry tag and BOTLINK_GATEWAY_ADDR is the main host's <ip>:9443.
|
||||||
|
name: scrabble-bot
|
||||||
|
|
||||||
|
services:
|
||||||
|
bot:
|
||||||
|
container_name: scrabble-telegram-bot
|
||||||
|
image: ${BOT_IMAGE:?set BOT_IMAGE to the registry tag}
|
||||||
|
restart: unless-stopped
|
||||||
|
logging:
|
||||||
|
driver: json-file
|
||||||
|
options:
|
||||||
|
max-size: "10m"
|
||||||
|
max-file: "3"
|
||||||
|
environment:
|
||||||
|
TELEGRAM_BOT_TOKEN: ${TELEGRAM_BOT_TOKEN:?set TELEGRAM_BOT_TOKEN}
|
||||||
|
TELEGRAM_GAME_CHANNEL_ID: ${TELEGRAM_GAME_CHANNEL_ID:-}
|
||||||
|
TELEGRAM_CHAT_ID: ${TELEGRAM_CHAT_ID:-}
|
||||||
|
TELEGRAM_PROMO_BOT_TOKEN: ${TELEGRAM_PROMO_BOT_TOKEN:-}
|
||||||
|
TELEGRAM_BOT_USERNAME: ${TELEGRAM_BOT_USERNAME:-}
|
||||||
|
TELEGRAM_BOT_LINK: ${TELEGRAM_BOT_LINK:-}
|
||||||
|
TELEGRAM_MINIAPP_URL: ${TELEGRAM_MINIAPP_URL:?set TELEGRAM_MINIAPP_URL}
|
||||||
|
# Real Bot API in prod (the test contour pins TELEGRAM_TEST_ENV=true instead).
|
||||||
|
TELEGRAM_TEST_ENV: "false"
|
||||||
|
TELEGRAM_API_BASE_URL: ${TELEGRAM_API_BASE_URL:-}
|
||||||
|
TELEGRAM_OWNS_UPDATES: "true"
|
||||||
|
# Dials the main host's published bot-link. ServerName stays `gateway` (the cert
|
||||||
|
# SAN), so TLS validation is independent of the dial address.
|
||||||
|
TELEGRAM_GATEWAY_ADDR: ${BOTLINK_GATEWAY_ADDR:?set BOTLINK_GATEWAY_ADDR (main:9443)}
|
||||||
|
TELEGRAM_BOTLINK_SERVER_NAME: gateway
|
||||||
|
TELEGRAM_BOTLINK_TLS_CERT: /certs/bot.crt
|
||||||
|
TELEGRAM_BOTLINK_TLS_KEY: /certs/bot.key
|
||||||
|
TELEGRAM_BOTLINK_TLS_CA: /certs/ca.crt
|
||||||
|
TELEGRAM_LOG_LEVEL: ${LOG_LEVEL:-info}
|
||||||
|
TELEGRAM_SERVICE_NAME: scrabble-telegram-bot
|
||||||
|
# No telemetry export: otelcol is on the main host, unreachable from here.
|
||||||
|
TELEGRAM_OTEL_TRACES_EXPORTER: none
|
||||||
|
TELEGRAM_OTEL_METRICS_EXPORTER: none
|
||||||
|
GOMAXPROCS: "1"
|
||||||
|
volumes:
|
||||||
|
- ${SCRABBLE_CONFIG_DIR:-.}/certs:/certs:ro
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
cpus: "1.0"
|
||||||
|
memory: 256M
|
||||||
@@ -0,0 +1,98 @@
|
|||||||
|
# Production main-host overlay, applied on top of docker-compose.yml on the main host:
|
||||||
|
# docker compose -f docker-compose.yml -f docker-compose.prod.yml up -d
|
||||||
|
#
|
||||||
|
# It (1) publishes caddy 80/443 — there is no host caddy in prod, so the contour caddy
|
||||||
|
# owns the edge and does its own ACME on CADDY_SITE_ADDRESS — and the gateway bot-link
|
||||||
|
# :9443 the remote bot dials in over mTLS; and (2) retunes the R7 limits down for the
|
||||||
|
# 2 vCPU / 1.9 GiB host (GOMAXPROCS=2, smaller memory caps, shorter Prometheus
|
||||||
|
# retention). The contour launches deliberately undersized at zero players; the added
|
||||||
|
# node_exporter + Grafana watch host memory so it can be resized at Selectel when
|
||||||
|
# traffic arrives.
|
||||||
|
#
|
||||||
|
# The bot + its VPN sidecar are absent here (the telegram-local profile is not
|
||||||
|
# activated); the prod bot runs on its own host from docker-compose.bot.yml.
|
||||||
|
|
||||||
|
services:
|
||||||
|
caddy:
|
||||||
|
ports:
|
||||||
|
- "80:80"
|
||||||
|
- "443:443"
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 96M
|
||||||
|
|
||||||
|
gateway:
|
||||||
|
# Prod pulls the pushed image by tag instead of building locally; the base
|
||||||
|
# build: section stays dormant because the deploy always pulls first.
|
||||||
|
image: ${REGISTRY:?set REGISTRY}/scrabble-gateway:${TAG:?set TAG}
|
||||||
|
ports:
|
||||||
|
- "9443:9443"
|
||||||
|
environment:
|
||||||
|
# 2 vCPU host: align the Go scheduler with the cgroup quota (R7's 3 needs 3 cores).
|
||||||
|
GOMAXPROCS: "2"
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
cpus: "2.0"
|
||||||
|
memory: 384M
|
||||||
|
|
||||||
|
backend:
|
||||||
|
image: ${REGISTRY:?set REGISTRY}/scrabble-backend:${TAG:?set TAG}
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 384M
|
||||||
|
|
||||||
|
postgres:
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 384M
|
||||||
|
|
||||||
|
validator:
|
||||||
|
image: ${REGISTRY:?set REGISTRY}/scrabble-telegram-validator:${TAG:?set TAG}
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 96M
|
||||||
|
|
||||||
|
landing:
|
||||||
|
image: ${REGISTRY:?set REGISTRY}/scrabble-landing:${TAG:?set TAG}
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 64M
|
||||||
|
|
||||||
|
otelcol:
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 256M
|
||||||
|
|
||||||
|
prometheus:
|
||||||
|
command:
|
||||||
|
- --config.file=/etc/prometheus/prometheus.yml
|
||||||
|
- --storage.tsdb.retention.time=7d
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 256M
|
||||||
|
|
||||||
|
tempo:
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 384M
|
||||||
|
|
||||||
|
grafana:
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 256M
|
||||||
|
|
||||||
|
postgres_exporter:
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 64M
|
||||||
@@ -71,6 +71,8 @@ services:
|
|||||||
# Seed dictionary for a FRESH volume; the per-contour value comes from the
|
# Seed dictionary for a FRESH volume; the per-contour value comes from the
|
||||||
# deploy env (Gitea TEST_/PROD_DICT_VERSION). See the volume note below.
|
# deploy env (Gitea TEST_/PROD_DICT_VERSION). See the volume note below.
|
||||||
DICT_VERSION: ${DICT_VERSION:-v1.2.1}
|
DICT_VERSION: ${DICT_VERSION:-v1.2.1}
|
||||||
|
# Build version stamped into the binary (git tag; see pkg/version).
|
||||||
|
VERSION: ${APP_VERSION:-dev}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
logging: *default-logging
|
logging: *default-logging
|
||||||
depends_on:
|
depends_on:
|
||||||
@@ -132,6 +134,8 @@ services:
|
|||||||
VITE_TELEGRAM_GAME_CHANNEL_NAME: ${VITE_TELEGRAM_GAME_CHANNEL_NAME:-}
|
VITE_TELEGRAM_GAME_CHANNEL_NAME: ${VITE_TELEGRAM_GAME_CHANNEL_NAME:-}
|
||||||
VITE_GATEWAY_URL: ${VITE_GATEWAY_URL:-}
|
VITE_GATEWAY_URL: ${VITE_GATEWAY_URL:-}
|
||||||
VITE_APP_VERSION: ${APP_VERSION:-dev}
|
VITE_APP_VERSION: ${APP_VERSION:-dev}
|
||||||
|
# Go binary version (the SPA's VITE_APP_VERSION is the same git tag).
|
||||||
|
VERSION: ${APP_VERSION:-dev}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
logging: *default-logging
|
logging: *default-logging
|
||||||
depends_on: [backend]
|
depends_on: [backend]
|
||||||
@@ -218,6 +222,8 @@ services:
|
|||||||
context: ..
|
context: ..
|
||||||
dockerfile: platform/telegram/Dockerfile
|
dockerfile: platform/telegram/Dockerfile
|
||||||
target: validator
|
target: validator
|
||||||
|
args:
|
||||||
|
VERSION: ${APP_VERSION:-dev}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
logging: *default-logging
|
logging: *default-logging
|
||||||
environment:
|
environment:
|
||||||
@@ -240,14 +246,22 @@ services:
|
|||||||
networks: [internal]
|
networks: [internal]
|
||||||
|
|
||||||
# --- Telegram bot (egress via the VPN sidecar in test; dials the gateway) ---
|
# --- Telegram bot (egress via the VPN sidecar in test; dials the gateway) ---
|
||||||
|
# vpn + bot are gated to the `telegram-local` profile: the test contour runs them
|
||||||
|
# locally (CI passes --profile telegram-local), the prod main host omits them, and
|
||||||
|
# the prod bot runs on its own host from deploy/docker-compose.bot.yml.
|
||||||
vpn:
|
vpn:
|
||||||
container_name: scrabble-telegram-vpn
|
container_name: scrabble-telegram-vpn
|
||||||
image: docker.iliadenisov.ru/developer/amneziawg-sidecar:latest
|
image: docker.iliadenisov.ru/developer/amneziawg-sidecar:latest
|
||||||
|
profiles: ["telegram-local"]
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
logging: *default-logging
|
logging: *default-logging
|
||||||
privileged: true
|
privileged: true
|
||||||
environment:
|
environment:
|
||||||
AWG_CONF: ${AWG_CONF:?set AWG_CONF}
|
# Required by the vpn sidecar, which is gated to the telegram-local profile.
|
||||||
|
# Compose can't scope a `:?` guard to a profile (interpolation runs for
|
||||||
|
# profiled-out services too) and the prod main host has no VPN, so this is a soft
|
||||||
|
# default; the test contour always supplies TEST_AWG_CONF and the sidecar validates it.
|
||||||
|
AWG_CONF: ${AWG_CONF:-}
|
||||||
networks:
|
networks:
|
||||||
internal:
|
internal:
|
||||||
aliases: [telegram]
|
aliases: [telegram]
|
||||||
@@ -255,10 +269,13 @@ services:
|
|||||||
bot:
|
bot:
|
||||||
container_name: scrabble-telegram-bot
|
container_name: scrabble-telegram-bot
|
||||||
image: scrabble-telegram-bot:latest
|
image: scrabble-telegram-bot:latest
|
||||||
|
profiles: ["telegram-local"]
|
||||||
build:
|
build:
|
||||||
context: ..
|
context: ..
|
||||||
dockerfile: platform/telegram/Dockerfile
|
dockerfile: platform/telegram/Dockerfile
|
||||||
target: bot
|
target: bot
|
||||||
|
args:
|
||||||
|
VERSION: ${APP_VERSION:-dev}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
logging: *default-logging
|
logging: *default-logging
|
||||||
depends_on: [vpn]
|
depends_on: [vpn]
|
||||||
@@ -444,6 +461,26 @@ services:
|
|||||||
memory: 128M
|
memory: 128M
|
||||||
networks: [internal]
|
networks: [internal]
|
||||||
|
|
||||||
|
# node_exporter exports host CPU/memory/disk metrics. The prod main host runs a tight
|
||||||
|
# 1.9 GiB budget, so host memory pressure — not just per-container docker_stats — is
|
||||||
|
# what warns before an OOM. Prometheus scrapes it at :9100 (see prometheus.yml).
|
||||||
|
node_exporter:
|
||||||
|
container_name: scrabble-node-exporter
|
||||||
|
image: quay.io/prometheus/node-exporter:v1.8.2
|
||||||
|
restart: unless-stopped
|
||||||
|
logging: *default-logging
|
||||||
|
command:
|
||||||
|
- --path.rootfs=/host
|
||||||
|
- --collector.filesystem.mount-points-exclude=^/(sys|proc|dev|host)($|/)
|
||||||
|
pid: host
|
||||||
|
volumes:
|
||||||
|
- /:/host:ro,rslave
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
limits:
|
||||||
|
memory: 64M
|
||||||
|
networks: [internal]
|
||||||
|
|
||||||
networks:
|
networks:
|
||||||
internal:
|
internal:
|
||||||
name: scrabble-internal
|
name: scrabble-internal
|
||||||
|
|||||||
Executable
+143
@@ -0,0 +1,143 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Production main-host deploy driver. Runs ON the main host, invoked over SSH by
|
||||||
|
# .gitea/workflows/prod-deploy.yaml as the deploy user (which must already be
|
||||||
|
# `docker login`ed to the registry). It pulls the images at the new tag and rolls
|
||||||
|
# the stack ONE service at a time in dependency order (least -> most dependent),
|
||||||
|
# health-checking after each; any failure rolls the whole stack back to the
|
||||||
|
# previously deployed tag.
|
||||||
|
#
|
||||||
|
# A schema migration adds a maintenance window: the backend (the only writer) is
|
||||||
|
# stopped so a consistent pg_dump is taken before the new backend migrates forward.
|
||||||
|
# Image rollback alone is safe under the expand-contract migration rule, so the
|
||||||
|
# automatic rollback never touches the database; the dump is kept for a MANUAL
|
||||||
|
# restore if a migration turned out to be destructive (see deploy/prod/README.md).
|
||||||
|
#
|
||||||
|
# Required env (exported by the workflow over SSH):
|
||||||
|
# REGISTRY registry namespace, e.g. docker.iliadenisov.ru/developer
|
||||||
|
# TAG new image tag (the deployed git SHA)
|
||||||
|
# PREV_TAG previously deployed tag, or "none" on the first deploy
|
||||||
|
# MIGRATION "1" when the deploy carries a schema migration, else "0"
|
||||||
|
# Optional: COMPOSE_DIR ENV_FILE DUMP_DIR STATE_FILE POSTGRES_USER POSTGRES_DB
|
||||||
|
set -uo pipefail
|
||||||
|
|
||||||
|
# Runtime compose vars (POSTGRES_*, GM_*, GRAFANA_*, CADDY_*, TELEGRAM_*, REGISTRY,
|
||||||
|
# SCRABBLE_CONFIG_DIR, ...) come from a shell-sourceable env file the workflow writes
|
||||||
|
# with single-quoted values. Exporting them into the process environment lets compose
|
||||||
|
# interpolate ${...} without re-parsing the value — a plain --env-file would mangle the
|
||||||
|
# literal '$' in the bcrypt GM_BASICAUTH_HASH.
|
||||||
|
ENV_FILE="${ENV_FILE:-/opt/scrabble/env.sh}"
|
||||||
|
# shellcheck disable=SC1090
|
||||||
|
[ -f "$ENV_FILE" ] && . "$ENV_FILE"
|
||||||
|
|
||||||
|
REGISTRY="${REGISTRY:?REGISTRY required (env.sh)}"
|
||||||
|
TAG="${TAG:?TAG required}"
|
||||||
|
PREV_TAG="${PREV_TAG:-none}"
|
||||||
|
MIGRATION="${MIGRATION:-0}"
|
||||||
|
COMPOSE_DIR="${COMPOSE_DIR:-/opt/scrabble/compose}"
|
||||||
|
DUMP_DIR="${DUMP_DIR:-/opt/scrabble/dumps}"
|
||||||
|
STATE_FILE="${STATE_FILE:-/opt/scrabble/DEPLOYED_TAG}"
|
||||||
|
# The prior deployed tag, preserved on every successful deploy so prod-rollback can
|
||||||
|
# target "the previous version" with no operator input.
|
||||||
|
PREV_STATE_FILE="${PREV_STATE_FILE:-/opt/scrabble/PREVIOUS_TAG}"
|
||||||
|
PG_USER="${POSTGRES_USER:-scrabble}"
|
||||||
|
PG_DB="${POSTGRES_DB:-scrabble}"
|
||||||
|
|
||||||
|
cd "$COMPOSE_DIR" || { echo "compose dir $COMPOSE_DIR missing"; exit 1; }
|
||||||
|
export REGISTRY
|
||||||
|
# otelcol joins the host docker group to read the socket; the GID varies per host.
|
||||||
|
DOCKER_GID="$(getent group docker | cut -d: -f3)"
|
||||||
|
export DOCKER_GID
|
||||||
|
|
||||||
|
dc() { docker compose -f docker-compose.yml -f docker-compose.prod.yml "$@"; }
|
||||||
|
use_tag() { export TAG="$1"; }
|
||||||
|
|
||||||
|
# --- health probes (one-off containers on the contour networks, like CI) --------
|
||||||
|
_probe() { docker run --rm --network "$1" alpine:3.20 wget -q -T 5 -O /dev/null "$2"; }
|
||||||
|
health_backend() { for _ in $(seq 1 20); do _probe scrabble-internal http://backend:8080/readyz && return 0; sleep 3; done; return 1; }
|
||||||
|
health_landing() { for _ in $(seq 1 20); do _probe scrabble-internal http://landing:80/ && return 0; sleep 3; done; return 1; }
|
||||||
|
health_postgres() { for _ in $(seq 1 30); do [ "$(docker inspect -f '{{.State.Health.Status}}' scrabble-postgres 2>/dev/null)" = healthy ] && return 0; sleep 2; done; return 1; }
|
||||||
|
health_running() { # health_running <container>: running, not restarting, stable restart count
|
||||||
|
local n="$1" s r c1 c2
|
||||||
|
for _ in $(seq 1 20); do
|
||||||
|
s="$(docker inspect -f '{{.State.Status}}' "$n" 2>/dev/null || echo missing)"
|
||||||
|
r="$(docker inspect -f '{{.State.Restarting}}' "$n" 2>/dev/null || echo true)"
|
||||||
|
if [ "$s" = running ] && [ "$r" = false ]; then
|
||||||
|
c1="$(docker inspect -f '{{.RestartCount}}' "$n")"; sleep 5
|
||||||
|
c2="$(docker inspect -f '{{.RestartCount}}' "$n")"
|
||||||
|
[ "$c1" = "$c2" ] && return 0
|
||||||
|
fi
|
||||||
|
sleep 3
|
||||||
|
done
|
||||||
|
return 1
|
||||||
|
}
|
||||||
|
|
||||||
|
roll() { # roll <service> <health-cmd...>
|
||||||
|
local svc="$1"; shift
|
||||||
|
echo ">>> rolling $svc -> $TAG"
|
||||||
|
dc up -d --no-build --no-deps "$svc" || return 1
|
||||||
|
"$@" || { echo "!!! $svc failed health check"; return 1; }
|
||||||
|
echo "<<< $svc healthy"
|
||||||
|
}
|
||||||
|
|
||||||
|
rollback() {
|
||||||
|
echo "########## ROLLBACK -> $PREV_TAG ##########"
|
||||||
|
if [ "$PREV_TAG" = none ]; then
|
||||||
|
echo "no previous tag (first deploy): cannot roll back; leaving the stack up for inspection."
|
||||||
|
return
|
||||||
|
fi
|
||||||
|
use_tag "$PREV_TAG"
|
||||||
|
dc up -d --no-build --remove-orphans
|
||||||
|
echo "rolled back to $PREV_TAG."
|
||||||
|
[ "$MIGRATION" = 1 ] && echo "NOTE: the DB is forward-migrated; a pre-deploy dump is in $DUMP_DIR — restore manually ONLY if the migration was destructive (see deploy/README.md, prod runbook)."
|
||||||
|
}
|
||||||
|
|
||||||
|
commit_tag() {
|
||||||
|
# Record the just-deployed tag as current, preserving the prior one as previous.
|
||||||
|
[ -f "$STATE_FILE" ] && cp "$STATE_FILE" "$PREV_STATE_FILE"
|
||||||
|
echo "$TAG" > "$STATE_FILE"
|
||||||
|
}
|
||||||
|
|
||||||
|
mkdir -p "$DUMP_DIR"
|
||||||
|
echo "=== prod deploy: tag=$TAG prev=$PREV_TAG migration=$MIGRATION ==="
|
||||||
|
use_tag "$TAG"
|
||||||
|
dc pull
|
||||||
|
|
||||||
|
# First deploy: nothing to roll from; bring the whole stack up and gate on health.
|
||||||
|
if [ -z "$(docker ps -aq -f name=scrabble-backend)" ]; then
|
||||||
|
echo "first deploy: bringing the whole stack up"
|
||||||
|
dc up -d --no-build --remove-orphans || { echo "compose up failed"; exit 1; }
|
||||||
|
health_backend || { echo "backend not ready"; exit 1; }
|
||||||
|
health_landing || { echo "landing not ready"; exit 1; }
|
||||||
|
commit_tag
|
||||||
|
echo "first deploy healthy ($TAG)."
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Migration deploy: freeze writes and snapshot a consistent dump before migrating.
|
||||||
|
if [ "$MIGRATION" = 1 ]; then
|
||||||
|
echo "migration deploy: opening maintenance window (stopping the backend = the only writer)"
|
||||||
|
dc stop backend
|
||||||
|
dump="$DUMP_DIR/pre-$TAG-$(date +%Y%m%d-%H%M%S).sql"
|
||||||
|
if ! docker exec scrabble-postgres pg_dump -U "$PG_USER" -d "$PG_DB" -n backend > "$dump"; then
|
||||||
|
echo "pg_dump failed; restarting the old backend and aborting"
|
||||||
|
dc start backend
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
echo "consistent dump: $dump"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Roll one service at a time, least -> most dependent; any failure rolls everything back.
|
||||||
|
roll postgres health_postgres || { rollback; exit 1; }
|
||||||
|
roll backend health_backend || { rollback; exit 1; }
|
||||||
|
roll gateway health_running scrabble-gateway || { rollback; exit 1; }
|
||||||
|
roll landing health_landing || { rollback; exit 1; }
|
||||||
|
roll validator health_running scrabble-telegram-validator || { rollback; exit 1; }
|
||||||
|
roll caddy health_running scrabble-caddy || { rollback; exit 1; }
|
||||||
|
|
||||||
|
# Observability + node_exporter: bring up the remainder and pick up any config changes.
|
||||||
|
dc up -d --no-build --remove-orphans || { rollback; exit 1; }
|
||||||
|
|
||||||
|
# Final internal sanity before committing the new tag.
|
||||||
|
health_backend || { rollback; exit 1; }
|
||||||
|
commit_tag
|
||||||
|
echo "=== deploy healthy ($TAG) ==="
|
||||||
@@ -18,3 +18,8 @@ scrape_configs:
|
|||||||
- job_name: postgres_exporter
|
- job_name: postgres_exporter
|
||||||
static_configs:
|
static_configs:
|
||||||
- targets: ["postgres_exporter:9187"]
|
- targets: ["postgres_exporter:9187"]
|
||||||
|
# Host-level metrics (memory/CPU/disk). Matters most on the prod main host's tight
|
||||||
|
# 1.9 GiB budget, where total host memory is the OOM-proximity signal.
|
||||||
|
- job_name: node
|
||||||
|
static_configs:
|
||||||
|
- targets: ["node_exporter:9100"]
|
||||||
|
|||||||
+39
-13
@@ -128,7 +128,11 @@ dropped). Horizontal scaling is explicit future work.
|
|||||||
and GCG are unaffected** (they stay decoded concrete characters, §9.1).
|
and GCG are unaffected** (they stay decoded concrete characters, §9.1).
|
||||||
- **gateway ↔ backend (sync)**: plain HTTP REST/JSON. The gateway injects
|
- **gateway ↔ backend (sync)**: plain HTTP REST/JSON. The gateway injects
|
||||||
`X-User-ID` for authenticated requests; `backend` never re-derives identity
|
`X-User-ID` for authenticated requests; `backend` never re-derives identity
|
||||||
from the body.
|
from the body. Because every sync call targets the one backend host, the
|
||||||
|
gateway's REST client widens its keep-alive pool well past the stdlib default
|
||||||
|
of 2 idle connections per host; otherwise the per-request connection churn
|
||||||
|
exhausts ephemeral ports and burns gateway CPU under load (see
|
||||||
|
[`../loadtest/REPORT.md`](../loadtest/REPORT.md)).
|
||||||
- **backend → gateway (live)**: a single gRPC server-stream carries live events
|
- **backend → gateway (live)**: a single gRPC server-stream carries live events
|
||||||
(your-turn, opponent-moved, chat, nudge). The gateway bridges them to the
|
(your-turn, opponent-moved, chat, nudge). The gateway bridges them to the
|
||||||
client's in-app stream while the app is open. Out-of-app delivery uses
|
client's in-app stream while the app is open. Out-of-app delivery uses
|
||||||
@@ -1052,8 +1056,10 @@ plaintext relay (`GATEWAY_BOTLINK_RELAY_ADDR`) the backend admin console calls.
|
|||||||
|
|
||||||
The full contour (`deploy/docker-compose.yml`) runs one `gateway`, one `backend`,
|
The full contour (`deploy/docker-compose.yml`) runs one `gateway`, one `backend`,
|
||||||
one Postgres, the static `landing`, the Telegram `validator` and `bot` (+ the bot's VPN
|
one Postgres, the static `landing`, the Telegram `validator` and `bot` (+ the bot's VPN
|
||||||
sidecar) and the **observability stack** —
|
sidecar — the `bot`+`vpn` pair is gated to a `telegram-local` compose profile so the prod
|
||||||
OTel Collector (OTLP/gRPC ingest → Prometheus metrics + Tempo traces) and Grafana
|
main host can omit them) and the **observability stack** —
|
||||||
|
OTel Collector (OTLP/gRPC ingest → Prometheus metrics + Tempo traces), a `node_exporter`
|
||||||
|
for host CPU/memory (the prod main host's OOM signal), and Grafana
|
||||||
with provisioned datasources and dashboards. All services export OTLP to the
|
with provisioned datasources and dashboards. All services export OTLP to the
|
||||||
collector; the bot shares the VPN sidecar's netns, so its `AWG_CONF` must not
|
collector; the bot shares the VPN sidecar's netns, so its `AWG_CONF` must not
|
||||||
carry a `DNS=` directive (that would hijack resolv.conf and stop it resolving
|
carry a `DNS=` directive (that would hijack resolv.conf and stop it resolving
|
||||||
@@ -1077,16 +1083,36 @@ Two contours, two secret/variable prefixes (`TEST_` / `PROD_`):
|
|||||||
generated by `deploy/gen-certs.sh` before `compose up`; the bot keeps its VPN sidecar
|
generated by `deploy/gen-certs.sh` before `compose up`; the bot keeps its VPN sidecar
|
||||||
for Telegram egress and dials the gateway by its internal name, so the bot-link stays
|
for Telegram egress and dials the gateway by its internal name, so the bot-link stays
|
||||||
on the internal network.
|
on the internal network.
|
||||||
- **Prod**: a manual SSH deploy after `development → master`. There is no
|
- **Prod**: a **manual** rollout — `.gitea/workflows/prod-deploy.yaml`, `workflow_dispatch`
|
||||||
host caddy, so the contour ships its own caddy terminating TLS — set
|
only (from `master`, `confirm=deploy`), run after `development → master` is merged green.
|
||||||
`CADDY_SITE_ADDRESS` to the domain and the caddy does its own ACME. The **bot runs
|
It builds and pushes the images to the registry (`docker.iliadenisov.ru`), then deploys
|
||||||
on a separate host** with native Telegram access (no VPN), deployed by SSH alongside
|
over SSH onto **two hosts** provisioned by `deploy/ansible/` (docker, a non-sudo `deploy`
|
||||||
the main app (rolled together so the bot-link protocol versions never skew); the
|
service account holding a dedicated CI key, key-only sshd, default-deny ufw, fail2ban):
|
||||||
gateway **publishes** the bot-link port and the certificates come from `PROD_`
|
the **main host** runs the full stack (`docker-compose.yml` + `docker-compose.prod.yml`),
|
||||||
secrets — a long-lived CA with leaves rotated by a scheduled job. The bot dials the
|
the **bot host** runs only the bot (`docker-compose.bot.yml`, no VPN — native Bot API
|
||||||
gateway's public bot-link endpoint and holds no inbound port; login is unaffected if
|
egress, telemetry off). There is no host caddy, so the contour caddy terminates TLS —
|
||||||
that host or the link is down. *(This prod wiring is the deferred final stage; the
|
`CADDY_SITE_ADDRESS` is the domain and caddy does its own ACME. The gateway **publishes**
|
||||||
code and the unified test contour land first — see `PRERELEASE.md`.)*
|
the bot-link `:9443`; the remote bot dials it over mTLS (certs from `PROD_BOTLINK_*`,
|
||||||
|
ServerName `gateway`, so TLS validation is independent of the public dial address), holds
|
||||||
|
no inbound port, and login is unaffected if that host or the link is down.
|
||||||
|
`deploy/prod-deploy.sh` rolls the main stack **one service at a time in dependency order**
|
||||||
|
(postgres → backend → gateway → landing → validator → caddy), health-checking after each;
|
||||||
|
any failure **rolls the whole stack back to the previous image tag**. A **schema migration**
|
||||||
|
adds a maintenance window: the backend (the sole writer) is stopped for a consistent
|
||||||
|
`pg_dump` before the new backend migrates forward — image rollback stays DB-safe under the
|
||||||
|
expand-contract migration rule, and the dump is kept for a manual restore. The workflow runs
|
||||||
|
four visible jobs (build → deploy-main → deploy-bot → verify). Releases are git tags
|
||||||
|
`vX.Y.Z`; the version is stamped into the image tag, every binary (`-ldflags` → `pkg/version`
|
||||||
|
→ the `service.version` telemetry attribute) and the SPA About screen. A separate manual
|
||||||
|
**`prod-rollback`** workflow re-deploys any prior release tag (blank input = the previous
|
||||||
|
deployed version, tracked on the host) over the same rolling, health-gated path — image-only,
|
||||||
|
no DB migration. The main host is
|
||||||
|
intentionally **launch-sized** (2 vCPU / 1.9 GiB): the prod overlay trims the R7 limits
|
||||||
|
(`GOMAXPROCS=2`, smaller caps, 7d Prometheus retention) and a **node_exporter** feeds
|
||||||
|
host-memory metrics to Grafana so it can be resized reactively as players arrive.
|
||||||
|
`GATEWAY_ABUSE_BAN_ENABLED=true` in prod (the per-IP ban is meaningful only with real
|
||||||
|
client IPs). The `vpn`+`bot` pair is gated to a `telegram-local` compose profile the test
|
||||||
|
contour activates; the prod main host omits it.
|
||||||
|
|
||||||
## 14. CI & branches
|
## 14. CI & branches
|
||||||
|
|
||||||
|
|||||||
+8
-1
@@ -31,7 +31,10 @@ ephemeral guest. The gateway validates the credential once and mints a thin
|
|||||||
session token; the backend resolves it to an internal `user_id`. A **Telegram Mini
|
session token; the backend resolves it to an internal `user_id`. A **Telegram Mini
|
||||||
App** launch authenticates from the platform's signed `initData`, themes the UI to
|
App** launch authenticates from the platform's signed `initData`, themes the UI to
|
||||||
the Telegram colours, and — on first contact — seeds the new account's interface
|
the Telegram colours, and — on first contact — seeds the new account's interface
|
||||||
language from the Telegram client. Telegram runs a **single bot**: every player uses
|
language from the Telegram client. If a launch cannot reach the backend (for example during a
|
||||||
|
deployment), the Mini App retries quietly and then shows a small "couldn't load" screen with a
|
||||||
|
**Retry** button, rather than dropping to the web sign-in, which has no place inside Telegram.
|
||||||
|
Telegram runs a **single bot**: every player uses
|
||||||
the same bot, and all of its chat and out-of-app notifications are written in the
|
the same bot, and all of its chat and out-of-app notifications are written in the
|
||||||
player's own **interface language** (en/ru). A separate optional **promo bot** can run alongside the
|
player's own **interface language** (en/ru). A separate optional **promo bot** can run alongside the
|
||||||
main one — its only job is to answer `/start` with a short message and a button that opens the
|
main one — its only job is to answer `/start` with a short message and a button that opens the
|
||||||
@@ -56,6 +59,10 @@ reconnect), and pending reads resume on their own — the interface stays usable
|
|||||||
flashing a red banner each time.
|
flashing a red banner each time.
|
||||||
|
|
||||||
### Accounts, linking & merge
|
### Accounts, linking & merge
|
||||||
|
_Sign-in is currently provider-only, so the in-profile linking UI is temporarily hidden; it
|
||||||
|
returns once the anonymous `/app/` guest (whose upgrade path this is) ships. The flow below
|
||||||
|
describes it for when it does._
|
||||||
|
|
||||||
First platform contact auto-provisions a durable account. From the profile a player
|
First platform contact auto-provisions a durable account. From the profile a player
|
||||||
links an email (via a confirm code) or their Telegram (via the web sign-in); a guest
|
links an email (via a confirm code) or their Telegram (via the web sign-in); a guest
|
||||||
who links their first identity becomes a durable account. The "already taken" status
|
who links their first identity becomes a durable account. The "already taken" status
|
||||||
|
|||||||
@@ -32,7 +32,10 @@ top-1 подсказку, безлимитную проверку слова с
|
|||||||
session-токен; backend сопоставляет его с внутренним `user_id`. Запуск **Telegram
|
session-токен; backend сопоставляет его с внутренним `user_id`. Запуск **Telegram
|
||||||
Mini App** авторизует по подписанным `initData` платформы, перекрашивает интерфейс
|
Mini App** авторизует по подписанным `initData` платформы, перекрашивает интерфейс
|
||||||
в цвета Telegram и — при первом контакте — задаёт язык интерфейса нового аккаунта по
|
в цвета Telegram и — при первом контакте — задаёт язык интерфейса нового аккаунта по
|
||||||
языку Telegram-клиента. Telegram держит **единого бота**: все игроки пользуются одним
|
языку Telegram-клиента. Если запуск не может достучаться до бэкенда (например, во время
|
||||||
|
деплоя), Mini App тихо повторяет попытки, а затем показывает небольшой экран «не удалось
|
||||||
|
загрузить» с кнопкой **Повторить**, вместо того чтобы сбрасывать на веб-вход, которому внутри
|
||||||
|
Telegram не место. Telegram держит **единого бота**: все игроки пользуются одним
|
||||||
и тем же ботом, а весь его чат и внеприложенческие уведомления пишутся на **языке
|
и тем же ботом, а весь его чат и внеприложенческие уведомления пишутся на **языке
|
||||||
интерфейса** самого игрока (en/ru). Рядом с основным может работать отдельный опциональный
|
интерфейса** самого игрока (en/ru). Рядом с основным может работать отдельный опциональный
|
||||||
**промо-бот** — его единственная задача отвечать на `/start` коротким сообщением и кнопкой,
|
**промо-бот** — его единственная задача отвечать на `/start` коротким сообщением и кнопкой,
|
||||||
@@ -57,6 +60,10 @@ Mini App** авторизует по подписанным `initData` плат
|
|||||||
рабочим вместо красного баннера каждый раз.
|
рабочим вместо красного баннера каждый раз.
|
||||||
|
|
||||||
### Аккаунты, привязка и слияние
|
### Аккаунты, привязка и слияние
|
||||||
|
_Вход сейчас только через провайдера, поэтому UI привязки в профиле временно скрыт; он
|
||||||
|
вернётся, когда появится анонимный `/app/`-гость (для апгрейда которого он и нужен). Описание
|
||||||
|
ниже — на этот случай._
|
||||||
|
|
||||||
Первый контакт с платформы заводит постоянный аккаунт. Из профиля игрок
|
Первый контакт с платформы заводит постоянный аккаунт. Из профиля игрок
|
||||||
привязывает email (по confirm-коду) или свой Telegram (через веб-вход); гость,
|
привязывает email (по confirm-коду) или свой Telegram (через веб-вход); гость,
|
||||||
привязавший первую личность, становится постоянным аккаунтом. Факт «личность уже
|
привязавший первую личность, становится постоянным аккаунтом. Факт «личность уже
|
||||||
|
|||||||
+3
-3
@@ -133,9 +133,9 @@ tests or touching CI.
|
|||||||
engine tests do). It is **not** part of the per-PR suite's behavioural assertions: it
|
engine tests do). It is **not** part of the per-PR suite's behavioural assertions: it
|
||||||
runs ad hoc as a one-shot container against the contour, producing a trip report (bugs
|
runs ad hoc as a one-shot container against the contour, producing a trip report (bugs
|
||||||
+ a per-container resource profile) read off the **otelcol `docker_stats` +
|
+ a per-container resource profile) read off the **otelcol `docker_stats` +
|
||||||
postgres_exporter** Grafana dashboard on the contour. Two passes are recorded — the
|
postgres_exporter** Grafana dashboard on the contour. The findings — including the
|
||||||
early [`REPORT-R2.md`](../loadtest/REPORT-R2.md) and the final, tuned
|
`game.evaluate` hot-path model and the gateway→backend connection-pool fix — are written
|
||||||
[`REPORT-R7.md`](../loadtest/REPORT-R7.md). See [`../loadtest/README.md`](../loadtest/README.md).
|
up in [`REPORT.md`](../loadtest/REPORT.md). See [`../loadtest/README.md`](../loadtest/README.md).
|
||||||
- **User feedback** — `internal/feedback` unit tests cover the attachment allow-list /
|
- **User feedback** — `internal/feedback` unit tests cover the attachment allow-list /
|
||||||
content-type and the channel normaliser; the UI covers `detectChannel`, the attachment gate and
|
content-type and the channel normaliser; the UI covers `detectChannel`, the attachment gate and
|
||||||
the feedback wire round-trip (`channel` / `feedback` / `codec` tests) plus a Playwright e2e
|
the feedback wire round-trip (`channel` / `feedback` / `codec` tests) plus a Playwright e2e
|
||||||
|
|||||||
+3
-1
@@ -70,7 +70,9 @@ RUN rm gateway/internal/webui/dist/landing.html
|
|||||||
# Reduce the workspace to what the gateway needs: gateway + pkg (loadtest is not in
|
# Reduce the workspace to what the gateway needs: gateway + pkg (loadtest is not in
|
||||||
# this context; its scrabble/gateway replace targets ./gateway, which is present here).
|
# this context; its scrabble/gateway replace targets ./gateway, which is present here).
|
||||||
RUN go work edit -dropuse=./backend -dropuse=./platform/telegram -dropuse=./loadtest
|
RUN go work edit -dropuse=./backend -dropuse=./platform/telegram -dropuse=./loadtest
|
||||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/gateway ./gateway/cmd/gateway
|
# VERSION (the deploy passes the git tag) is stamped into the binary via the linker.
|
||||||
|
ARG VERSION=dev
|
||||||
|
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/gateway ./gateway/cmd/gateway
|
||||||
|
|
||||||
# --- runtime -----------------------------------------------------------------
|
# --- runtime -----------------------------------------------------------------
|
||||||
FROM gcr.io/distroless/static-debian12:nonroot AS gateway
|
FROM gcr.io/distroless/static-debian12:nonroot AS gateway
|
||||||
|
|||||||
@@ -22,6 +22,19 @@ import (
|
|||||||
pushv1 "scrabble/pkg/proto/push/v1"
|
pushv1 "scrabble/pkg/proto/push/v1"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
// backendMaxIdleConns sizes the REST keep-alive pool to the single backend host. The
|
||||||
|
// default transport caps idle connections per host at 2 (http.DefaultMaxIdleConnsPerHost),
|
||||||
|
// which — since every synchronous client call proxies to that one host — forces a fresh
|
||||||
|
// TCP connection (and a lingering TIME_WAIT socket) for almost every request under load.
|
||||||
|
// That connection churn burns gateway CPU and exhausts ephemeral ports at scale, all
|
||||||
|
// while the backend itself sits near-idle. Pooling the connections lets them be reused.
|
||||||
|
//
|
||||||
|
// The stress harness measured the effect at 500 concurrent players: the churn collapsed
|
||||||
|
// from ~26 500 TIME_WAIT sockets to ~0 and peak gateway CPU from ~1.75 to ~0.26 cores,
|
||||||
|
// with the pool settling at ~225 live connections. 512 keeps ~2x headroom over that
|
||||||
|
// observed peak so a burst never re-caps the pool. See loadtest/REPORT.md.
|
||||||
|
const backendMaxIdleConns = 512
|
||||||
|
|
||||||
// Client calls the backend's REST API and opens its push gRPC stream.
|
// Client calls the backend's REST API and opens its push gRPC stream.
|
||||||
type Client struct {
|
type Client struct {
|
||||||
baseURL string
|
baseURL string
|
||||||
@@ -41,9 +54,14 @@ func New(httpURL, grpcAddr string, timeout time.Duration) (*Client, error) {
|
|||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("backendclient: dial push %s: %w", grpcAddr, err)
|
return nil, fmt.Errorf("backendclient: dial push %s: %w", grpcAddr, err)
|
||||||
}
|
}
|
||||||
|
// Clone the default transport (keeping its proxy, dialer and timeouts) and widen the
|
||||||
|
// idle pool so REST calls to the backend reuse connections instead of churning them.
|
||||||
|
transport := http.DefaultTransport.(*http.Transport).Clone()
|
||||||
|
transport.MaxIdleConns = backendMaxIdleConns
|
||||||
|
transport.MaxIdleConnsPerHost = backendMaxIdleConns
|
||||||
return &Client{
|
return &Client{
|
||||||
baseURL: strings.TrimRight(httpURL, "/"),
|
baseURL: strings.TrimRight(httpURL, "/"),
|
||||||
http: &http.Client{Timeout: timeout},
|
http: &http.Client{Timeout: timeout, Transport: transport},
|
||||||
conn: conn,
|
conn: conn,
|
||||||
push: pushv1.NewPushClient(conn),
|
push: pushv1.NewPushClient(conn),
|
||||||
}, nil
|
}, nil
|
||||||
|
|||||||
@@ -0,0 +1,31 @@
|
|||||||
|
package backendclient
|
||||||
|
|
||||||
|
import (
|
||||||
|
"net/http"
|
||||||
|
"testing"
|
||||||
|
"time"
|
||||||
|
)
|
||||||
|
|
||||||
|
// TestBackendTransportPoolsConnections guards the fix for the gateway->backend
|
||||||
|
// connection churn. Every synchronous client call proxies to the single backend host,
|
||||||
|
// so the REST client must widen the idle-connection pool past the default per-host cap
|
||||||
|
// of 2 (http.DefaultMaxIdleConnsPerHost) — otherwise almost every request under load
|
||||||
|
// opens a fresh TCP connection that then lingers in TIME_WAIT, burning gateway CPU and
|
||||||
|
// exhausting ephemeral ports. Reverting to the default transport (`&http.Client{...}`
|
||||||
|
// with no Transport) would silently reintroduce that, so assert the pool is widened.
|
||||||
|
func TestBackendTransportPoolsConnections(t *testing.T) {
|
||||||
|
c, err := New("http://backend.invalid", "localhost:9090", time.Second)
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("New: %v", err)
|
||||||
|
}
|
||||||
|
defer func() { _ = c.Close() }()
|
||||||
|
|
||||||
|
tr, ok := c.http.Transport.(*http.Transport)
|
||||||
|
if !ok {
|
||||||
|
t.Fatalf("REST transport = %T, want a *http.Transport with a widened idle pool", c.http.Transport)
|
||||||
|
}
|
||||||
|
if tr.MaxIdleConnsPerHost <= http.DefaultMaxIdleConnsPerHost {
|
||||||
|
t.Errorf("MaxIdleConnsPerHost = %d, want > default %d (else per-call connection churn)",
|
||||||
|
tr.MaxIdleConnsPerHost, http.DefaultMaxIdleConnsPerHost)
|
||||||
|
}
|
||||||
|
}
|
||||||
+12
-8
@@ -15,10 +15,12 @@ and prints a trip-report summary. It stays in the repo for repeats.
|
|||||||
2. **Drive** (edge protocol over h2c): assembles real 2–4 player games via the
|
2. **Drive** (edge protocol over h2c): assembles real 2–4 player games via the
|
||||||
invitation flow (`invitation.create` → `invitation.accept`, no robots), then runs
|
invitation flow (`invitation.create` → `invitation.accept`, no robots), then runs
|
||||||
each player's turn loop — poll `game.state`, replay `game.history`, generate a legal
|
each player's turn loop — poll `game.state`, replay `game.history`, generate a legal
|
||||||
**mid-ranked** move with the embedded `scrabble-solver`, and `game.submit_play`
|
**mid-ranked** move with the embedded `scrabble-solver`, **compose it tile by tile with
|
||||||
(or pass/exchange). A fraction of turns exercise nudge / chat / check-word / draft /
|
the debounced `game.evaluate` preview a real client fires** (the hottest gameplay call),
|
||||||
profile-update / stats. Each player also holds a live `Subscribe` stream. The
|
persist a `draft.save`, and `game.submit_play` (or pass/exchange). A fraction of turns
|
||||||
moderate ramp is **50 → 200 → 500** concurrent players, ~12 min per step.
|
exercise nudge / chat / check-word / draft / profile-update / stats. Each player also
|
||||||
|
holds a live `Subscribe` stream. The moderate ramp is **50 → 200 → 500** concurrent
|
||||||
|
players, ~12 min per step. `--eval=false` drops the evaluate model for an A/B baseline.
|
||||||
3. **Hammer**: drives `games.list` from one account far above the per-user rate limit
|
3. **Hammer**: drives `games.list` from one account far above the per-user rate limit
|
||||||
to verify the limiter holds (`rate_limited` results) and measure its cost.
|
to verify the limiter holds (`rate_limited` results) and measure its cost.
|
||||||
4. **Report**: per-operation latency percentiles, throughput, result-code breakdown,
|
4. **Report**: per-operation latency percentiles, throughput, result-code breakdown,
|
||||||
@@ -72,6 +74,8 @@ Key `run` flags (env in parentheses):
|
|||||||
| `--games-per-player` | `0` (random 3–5) | target concurrent games per player |
|
| `--games-per-player` | `0` (random 3–5) | target concurrent games per player |
|
||||||
| `--tick` | `800ms` | per-player op cadence (keeps a player under the per-user limit) |
|
| `--tick` | `800ms` | per-player op cadence (keeps a player under the per-user limit) |
|
||||||
| `--secondary-prob` | `0.08` | chance per tick of a non-move op |
|
| `--secondary-prob` | `0.08` | chance per tick of a non-move op |
|
||||||
|
| `--eval` | `true` | model the per-tile `game.evaluate` preview (the gameplay hot path); `false` reproduces the pre-evaluate harness |
|
||||||
|
| `--eval-recon` | `1` | extra full-composition evaluate re-previews per play (reconsideration), beyond one per placed tile |
|
||||||
| `--hammer-workers` / `--hammer-dur` | `20` / `15s` | gateway-hammer (0 workers disables) |
|
| `--hammer-workers` / `--hammer-dur` | `20` / `15s` | gateway-hammer (0 workers disables) |
|
||||||
| `--reset` / `--cleanup` | `false` | delete harness rows before / after the run |
|
| `--reset` / `--cleanup` | `false` | delete harness rows before / after the run |
|
||||||
|
|
||||||
@@ -93,11 +97,11 @@ runs unconditionally. Use an **absolute** path (here via `$PWD`): `go test ./loa
|
|||||||
runs each package from its own directory, so a relative `BACKEND_DICT_DIR` would not
|
runs each package from its own directory, so a relative `BACKEND_DICT_DIR` would not
|
||||||
resolve.
|
resolve.
|
||||||
|
|
||||||
## Trip reports
|
## Trip report
|
||||||
|
|
||||||
The two stress passes are written up in the repo: the early pass in
|
The stress findings — the final run, the `game.evaluate` hot-path model, the
|
||||||
[`REPORT-R2.md`](REPORT-R2.md) and the final, tuned pass in
|
gateway→backend connection-pool fix, and the revised sizing — are written up in
|
||||||
[`REPORT-R7.md`](REPORT-R7.md).
|
[`REPORT.md`](REPORT.md).
|
||||||
|
|
||||||
## Caveat
|
## Caveat
|
||||||
|
|
||||||
|
|||||||
@@ -1,162 +0,0 @@
|
|||||||
# R2 — early stress-run trip report
|
|
||||||
|
|
||||||
The early stress pass for `PRERELEASE.md` R2. It exercises the system through the
|
|
||||||
**edge protocol** with the `scrabble/loadtest` harness, to surface logic/concurrency
|
|
||||||
bugs and capture a resource baseline that feeds R3 (edge hardening), R6 (refactor) and
|
|
||||||
R7 (final tuning). Pass bar: **diagnostic** — the run "passes" by completing without the
|
|
||||||
harness crashing; findings are recorded below, not gated.
|
|
||||||
|
|
||||||
## Method
|
|
||||||
|
|
||||||
- **Driver:** the `scrabble/loadtest` module, run as a one-shot container on the
|
|
||||||
`scrabble-internal` docker network (reaching `postgres:5432` and `gateway:8081`
|
|
||||||
directly, bypassing the host→gateway hairpin).
|
|
||||||
- **Seed:** 10 000 durable + 1 000 guest accounts with pre-created sessions written
|
|
||||||
directly to Postgres (token hash matches `backend/internal/session`), so the driver
|
|
||||||
authenticates without the per-IP-limited auth ops.
|
|
||||||
- **Games:** assembled through the real **invitation** flow (`invitation.create` →
|
|
||||||
`invitation.accept`), 2–4 players each, no robots; variants spread over
|
|
||||||
scrabble_en / scrabble_ru / erudit_ru.
|
|
||||||
- **Play:** each virtual player holds a live `Subscribe` stream and, per tick, polls
|
|
||||||
`game.state`, replays `game.history` and submits a **mid-ranked** legal move generated
|
|
||||||
locally by the embedded `scrabble-solver` (the edge carries no board), or
|
|
||||||
passes/exchanges; a fraction exercise nudge / chat / check-word / draft / profile /
|
|
||||||
stats. A separate **gateway-hammer** floods `games.list` from one account.
|
|
||||||
- **Scale:** moderate ramp **50 → 200 → 500** concurrent players, 10 min/step (the
|
|
||||||
agreed moderate profile; harness and contour share this host's CPU).
|
|
||||||
- **Resource capture:** `docker stats` (docker API) sampled every 28 s for per-container
|
|
||||||
CPU/memory; Prometheus for edge latency/throughput, `postgres_exporter` internals and
|
|
||||||
per-service Go runtime metrics.
|
|
||||||
|
|
||||||
## Run configuration
|
|
||||||
|
|
||||||
```
|
|
||||||
loadtest run --durable 10000 --guest 1000 --steps 50,200,500 --step-dur 10m \
|
|
||||||
--tick 800ms --hammer-workers 20 --hammer-dur 15s --cleanup
|
|
||||||
```
|
|
||||||
|
|
||||||
Date: 2026-06-09. Contour: the R1-baseline schema, freshly deployed with the R2
|
|
||||||
exporters. Seeded population removed by `--cleanup` afterwards.
|
|
||||||
|
|
||||||
## Findings
|
|
||||||
|
|
||||||
### Validated (fixed within R2)
|
|
||||||
- **Harness draft payload.** `draft.save` first returned `bad_request`: the backend
|
|
||||||
draft DTO's `rack_order` is a string (the harness sent `[]`). Fixed → `ok`.
|
|
||||||
- **Harness profile marker.** `profile.update` first returned `invalid_profile`: the
|
|
||||||
editable-display-name validator (`backend/internal/account/profile.go`) forbids digits
|
|
||||||
and colons, but the seed marker was `lt:…`. Switched the marker to a distinctive
|
|
||||||
letters-only string → `ok`. Cleanup still matches it.
|
|
||||||
|
|
||||||
### By-design behaviour (correctly exercised, not bugs)
|
|
||||||
- **`chat_not_your_turn`** — chat is gated to the sender's turn
|
|
||||||
(`backend/internal/social/chat.go`); off-turn posts are correctly rejected.
|
|
||||||
- **`nudge_own_turn`** — you nudge the player whose turn it is, so a nudge on your own
|
|
||||||
turn is correctly rejected. The harness nudges/chats at random ticks, so a share of
|
|
||||||
these codes is expected.
|
|
||||||
|
|
||||||
### Observability gap (key R7 input)
|
|
||||||
- **cAdvisor yields only the root cgroup on the contour host.** Its docker factory
|
|
||||||
registers, but per-container init fails — `failed to identify the read-write layer ID
|
|
||||||
… /rootfs/var/lib/docker/image/overlayfs/…: no such file or directory` — because this
|
|
||||||
host's `/var/lib/docker` is a **separate XFS mount** not visible under cAdvisor's
|
|
||||||
`/rootfs` bind (the existing galaxy deployment on the same host has the same
|
|
||||||
limitation). So the **Scrabble — Resources** dashboard's per-container panels are empty
|
|
||||||
here, and per-container CPU/RSS for this run was captured via `docker stats` instead.
|
|
||||||
Postgres internals (`postgres_exporter`) and per-service Go runtime metrics
|
|
||||||
(`go_*` by `service_name`) work. **Recommendation for R7:** adopt the otelcol
|
|
||||||
**`docker_stats`** receiver (already the contrib image) — it reads per-container stats
|
|
||||||
via the docker API with no cgroup dependency — and/or run the final pass on hardware
|
|
||||||
where cAdvisor resolves containers. (Decision to confirm with the owner.)
|
|
||||||
|
|
||||||
### Run results
|
|
||||||
|
|
||||||
The ramp ran clean to 500 players with no harness crash, no deadlock and
|
|
||||||
`stream errors: 0`; cleanup removed all 11 000 seeded accounts (and their ~941 games).
|
|
||||||
|
|
||||||
- **Ramp:** step 1 = 50 players / 90 games, step 2 = 200 / 282, step 3 = 500 / 569.
|
|
||||||
- **Volume (30 min):** 1.20 M total edge calls, 659 req/s average. Real gameplay at
|
|
||||||
scale: **48 870 committed plays**, 52 772 `your_turn` + 159 631 `opponent_moved`
|
|
||||||
events, **2 798 games finished**.
|
|
||||||
- **Latency under load (peak, step 3):** `game.state` p50 ≈ 100 ms, p90/p99 in the
|
|
||||||
200–500 ms buckets, max 849 ms; `game.submit_play` similar (p99 ≤ 500 ms, max 490 ms).
|
|
||||||
Lobby ops stayed fast (invitation/games.list p99 ≤ 10 ms).
|
|
||||||
- **Rate limiter holds.** The gateway-hammer sent 522 667 `games.list` from one account;
|
|
||||||
**522 486 (99.97 %) were `rate_limited`**, only 135 `ok` (the burst). Rejections are
|
|
||||||
cheap — p99 = 2 ms — and the gateway sustained ~16 k req/s of rejections during the
|
|
||||||
flood. The per-user limiter behaves as designed (R3 input: the cost is negligible).
|
|
||||||
|
|
||||||
**Top finding — `transport_error` under saturation.** At 500 players ~14 % of
|
|
||||||
`game.state` calls (72 429 / 519 067) and a few % of the other ops returned a Connect
|
|
||||||
`transport_error` (not a domain code). It correlates with the CPU saturation below: the
|
|
||||||
backend/gateway are pinned near one core each while the host also runs the 86 %-core
|
|
||||||
harness, so the edge sheds load (resets/timeouts) at the knee. It is **amplified by a
|
|
||||||
harness artifact** — all 500 virtual players multiplex over a *single* shared
|
|
||||||
`http2.Transport`, so 500 persistent `Subscribe` streams plus Execute calls press on one
|
|
||||||
HTTP/2 connection's concurrent-stream limit; real clients each use their own connection.
|
|
||||||
**Actions:** R7 harness — give each player (or a pool) its own transport, and run on
|
|
||||||
hardware not shared with the contour; R3 — confirm the gateway's h2c
|
|
||||||
`MaxConcurrentStreams` and edge timeouts are sized for many persistent streams.
|
|
||||||
|
|
||||||
**Minor findings:**
|
|
||||||
- `unauthenticated` on a tiny share (188 / 519 067 `game.state`, ~0.04 %) — transient
|
|
||||||
session-resolve failures under load; worth a glance in R3 but not material.
|
|
||||||
- one `internal` on `game.pass` (1 / 4 788).
|
|
||||||
- `game_finished` dominates `chat.nudge`/`chat.post` (≈ 3 900 each): the harness keeps
|
|
||||||
secondary ops on games that already ended. Harness refinement — drop finished games
|
|
||||||
from the rotation (R7).
|
|
||||||
- `nudge_own_turn` / `chat_not_your_turn` / `nudge_too_soon` are the expected turn/rate
|
|
||||||
gates, correctly exercised.
|
|
||||||
|
|
||||||
## Resource baseline
|
|
||||||
|
|
||||||
Per-container peak during step 3 (500 players), from `docker stats`:
|
|
||||||
|
|
||||||
| container | peak CPU | memory |
|
|
||||||
|-----------|---------:|-------:|
|
|
||||||
| scrabble-backend | **99 %** (~1 core) | 91 MiB |
|
|
||||||
| scrabble-gateway | **93 %** | 76 MiB |
|
|
||||||
| scrabble-postgres | **90 %** | 69 MiB |
|
|
||||||
| scrabble-loadtest (harness) | **86 %** | 42 MiB |
|
|
||||||
| scrabble-otelcol | 10 % | 110 MiB |
|
|
||||||
| scrabble-tempo | 9 % | 446 MiB |
|
|
||||||
| prometheus / postgres-exporter | ~0 % | 46 / 16 MiB |
|
|
||||||
|
|
||||||
- **The contour is CPU-bound at 500 concurrent players:** backend, gateway and Postgres
|
|
||||||
each saturate ~1 core (single-instance MVP config), so the system draws ~3 cores at
|
|
||||||
this scale; memory is modest (≤ 100 MiB per Go service). This is the sizing input for
|
|
||||||
R7 (pool sizes, GOMAXPROCS, container limits) and the prod cutover.
|
|
||||||
- **Caveat:** the harness itself peaked at **86 % of a core** on the *same host*, so the
|
|
||||||
step-3 latency and `transport_error` figures are pessimistic — the contour competed
|
|
||||||
with the generator for CPU. A clean ceiling needs separate hardware (R7).
|
|
||||||
- **Postgres:** peak 28 backend connections, ~5 581 commits/s at the peak, **100 % cache
|
|
||||||
hit ratio** (no disk reads) — the DB was comfortable; CPU, not I/O, is its limit here.
|
|
||||||
- **Goroutines:** backend 638, gateway **1 698** (it holds the 500 `Subscribe` streams +
|
|
||||||
per-request goroutines), telegram 49 — all stable, no leak across the ramp.
|
|
||||||
|
|
||||||
## Recommendations feeding later phases
|
|
||||||
- **R3 (edge hardening):** the per-user limiter holds (99.97 % rejected, p99 2 ms) — add
|
|
||||||
the per-IP body-size cap on top. Investigate the **~14 % `transport_error` on
|
|
||||||
`game.state` at 500 players**: confirm the gateway h2c `MaxConcurrentStreams` and edge
|
|
||||||
read/write timeouts are sized for many persistent `Subscribe` streams, and glance at the
|
|
||||||
~0.04 % transient `unauthenticated` resolves under load.
|
|
||||||
- **R6 (refactor):** no logic bug forced a code change beyond the two harness-payload
|
|
||||||
fixes; the run surfaced no deadlock or goroutine leak across the ramp.
|
|
||||||
- **R7 (final tuning + stress):** (1) fix the per-container observability gap — adopt the
|
|
||||||
otelcol `docker_stats` receiver so Grafana shows per-container CPU/RSS on the contour;
|
|
||||||
(2) refine the harness — per-player/pooled transports and dropping finished games from
|
|
||||||
the rotation — and run on hardware **not** shared with the contour; (3) size pools /
|
|
||||||
GOMAXPROCS / container limits from the CPU-bound peak (~1 core each for backend, gateway,
|
|
||||||
Postgres at 500 players).
|
|
||||||
|
|
||||||
## Re-running
|
|
||||||
|
|
||||||
See [`README.md`](README.md). Briefly, from the repo root:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
|
||||||
docker run --rm --name scrabble-loadtest --network scrabble-internal \
|
|
||||||
-e POSTGRES_PASSWORD=… scrabble-loadtest run # add --reset on a re-run
|
|
||||||
```
|
|
||||||
|
|
||||||
The harness stays in the repo for the R7 repeat.
|
|
||||||
@@ -1,212 +0,0 @@
|
|||||||
# R7 — final stress-run trip report
|
|
||||||
|
|
||||||
The final pre-release stress pass for [`PRERELEASE.md`](../PRERELEASE.md) R7. It re-runs
|
|
||||||
the R2 harness (`scrabble/loadtest`) against the **final, refactored system** on a
|
|
||||||
freshly redeployed contour, to confirm the system holds at scale and to settle the
|
|
||||||
resource sizing (container limits, `GOMAXPROCS`, pools, rate limits, log levels) before
|
|
||||||
the Stage 18 prod cutover. Pass bar: **diagnostic + a tuning decision** — the run
|
|
||||||
"passes" by completing cleanly; the per-container resource profile drives the tuning
|
|
||||||
recorded below. Companion to the early pass, [`REPORT-R2.md`](REPORT-R2.md).
|
|
||||||
|
|
||||||
## What changed since the R2 pass
|
|
||||||
|
|
||||||
- **Harness — per-player transports.** Each virtual player now owns its `edge.Client`
|
|
||||||
(its own `http2.Transport` / h2c connection carrying both its `Subscribe` stream and
|
|
||||||
its `Execute` calls), instead of all players multiplexing over one shared transport.
|
|
||||||
R2 traced the ~14 % `transport_error` on `game.state` at 500 players to that single
|
|
||||||
shared connection's stream limit; per-player connections mirror real clients and
|
|
||||||
remove the artifact, so this pass measures the system, not the harness.
|
|
||||||
- **Harness — drop finished games.** `playTurn` reports a finished game and the player
|
|
||||||
drops it from its rotation, so secondary ops stop hitting `game_finished` on ended
|
|
||||||
games (the other R2 harness finding).
|
|
||||||
- **Observability — otelcol `docker_stats`.** cAdvisor (which resolves only the root
|
|
||||||
cgroup on this host — separate-XFS `/var/lib/docker`) is replaced by the otelcol
|
|
||||||
`docker_stats` receiver, reading per-container CPU/memory/network from the Docker API.
|
|
||||||
Per-container panels now populate on the contour host. (`api_version` pinned to 1.44;
|
|
||||||
the daemon's minimum is 1.40.)
|
|
||||||
- **Contour — container limits + `GOMAXPROCS`.** `deploy.resources.limits` now bound
|
|
||||||
every service; the Go services pin `GOMAXPROCS` to their CPU limit so the runtime
|
|
||||||
matches the cgroup quota. Starting values were generous over the R2 peak; this pass
|
|
||||||
validates them and settles the agreed sizing (below).
|
|
||||||
|
|
||||||
## Method
|
|
||||||
|
|
||||||
Unchanged from R2 except for the per-player transports and the dropped-finished-games
|
|
||||||
refinement above:
|
|
||||||
|
|
||||||
- **Driver:** the `scrabble/loadtest` module, run as a one-shot container on the
|
|
||||||
`scrabble-internal` docker network (reaching `postgres:5432` / `gateway:8081`
|
|
||||||
directly), capped at `--cpus 3` so the contour keeps the host's spare cores.
|
|
||||||
- **Seed:** 10 000 durable + 1 000 guest accounts with pre-created sessions written
|
|
||||||
straight to Postgres (token hash matches `backend/internal/session`).
|
|
||||||
- **Games:** assembled through the real **invitation** flow, 2–4 players each, no
|
|
||||||
robots; variants over scrabble_en / scrabble_ru / erudit_ru.
|
|
||||||
- **Play:** each player holds a live `Subscribe` stream and, per tick, polls
|
|
||||||
`game.state`, replays `game.history` and submits a **mid-ranked** legal move generated
|
|
||||||
locally by the embedded `scrabble-solver`, or passes / exchanges; a fraction exercise
|
|
||||||
nudge / chat / check-word / draft / profile / stats. A separate **gateway-hammer**
|
|
||||||
floods `games.list` from one account.
|
|
||||||
- **Scale:** the same moderate ramp **50 → 200 → 500** concurrent players, 10 min/step.
|
|
||||||
- **Resource capture:** `docker stats` (docker API) sampled every ~20 s for per-container
|
|
||||||
CPU/memory; the otelcol **`docker_stats`** receiver → Prometheus → the Grafana
|
|
||||||
**Scrabble — Resources** dashboard for the same per-container series; `postgres_exporter`
|
|
||||||
internals and per-service Go runtime metrics.
|
|
||||||
|
|
||||||
## Run configuration
|
|
||||||
|
|
||||||
```
|
|
||||||
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
|
||||||
-e POSTGRES_PASSWORD=… scrabble-loadtest \
|
|
||||||
run --durable 10000 --guest 1000 --steps 50,200,500 --step-dur 10m \
|
|
||||||
--tick 800ms --hammer-workers 20 --hammer-dur 15s --reset --cleanup
|
|
||||||
```
|
|
||||||
|
|
||||||
Date: 2026-06-10. Contour: the R1-baseline schema, freshly redeployed with the R7
|
|
||||||
container limits / `GOMAXPROCS` (backend/gateway/postgres capped at 2 cores + 512 MiB,
|
|
||||||
`GOMAXPROCS=2`) and the `docker_stats` observability. Seeded population removed by
|
|
||||||
`--cleanup` afterwards.
|
|
||||||
|
|
||||||
## Findings
|
|
||||||
|
|
||||||
The ramp ran clean to 500 players — no harness crash, no deadlock, `stream errors: 0` —
|
|
||||||
and cleanup removed all 11 000 seeded accounts.
|
|
||||||
|
|
||||||
- **Volume (1827 s):** 821 680 edge calls (449.7 req/s incl. the hammer). Real gameplay
|
|
||||||
at scale: **50 916 committed plays**, 4 817 passes, 2 931 games finished; 165 755
|
|
||||||
`opponent_moved` + 54 864 `your_turn` events.
|
|
||||||
- **The per-player transport fix worked.** `game.state` returned `transport_error` on
|
|
||||||
**3 173 / 127 403 = 2.49 %** of calls — down from R2's ~14 % on the same step. Other
|
|
||||||
ops were lower still (`game.history` 0.43 %, `game.submit_play` 0.28 %). The residual
|
|
||||||
is the gateway bursting into its 2-core cap (see the profile below), not the harness.
|
|
||||||
- **Dropping finished games worked.** `game_finished` on `chat.nudge` / `chat.post` fell
|
|
||||||
to **35 / 36** (R2: ≈ 3 900 each) — secondary ops no longer hammer ended games.
|
|
||||||
- **The limiter holds.** The gateway-hammer sent 565 152 `games.list`; **564 979
|
|
||||||
(99.97 %) were `rate_limited`** (154 ok burst, 19 deadline), p99 = 2 ms, ~309 req/s of
|
|
||||||
rejections sustained — unchanged from R2.
|
|
||||||
- **Latency (peak):** `game.state` p50 ≈ 100 ms, p99 in the 2000 ms bucket (max 2549 ms);
|
|
||||||
`game.submit_play` p50 100 / p99 1000 ms bucket. Lobby ops stayed fast
|
|
||||||
(invitation / games.list p99 ≤ 10 ms). The p99 tail correlates with the gateway
|
|
||||||
burst-throttling, not the backend (which stayed at ~0.85 core).
|
|
||||||
|
|
||||||
## Resource profile
|
|
||||||
|
|
||||||
Per-container peak during step 3 (500 players), with the R7 starting limits in force
|
|
||||||
(backend/gateway/postgres capped at 2 cores / 512 MiB). Two CPU columns: `docker stats`
|
|
||||||
samples a ~1 s window (catches bursts); the otelcol `docker_stats` receiver averages over
|
|
||||||
its 30 s collection interval (smooths them) — they agree within sampling error, which
|
|
||||||
validates the new observability path.
|
|
||||||
|
|
||||||
| container | CPU burst (1 s) | CPU sustained (30 s) | CPU cap | mem peak | mem cap |
|
|
||||||
|-----------|----------------:|---------------------:|--------:|---------:|--------:|
|
|
||||||
| scrabble-gateway | **217 %** (at cap) | ~145 % | 200 % | 167 MiB | 512 MiB |
|
|
||||||
| scrabble-postgres | 138 % | ~153 % | 200 % | 117 MiB | 512 MiB |
|
|
||||||
| scrabble-backend | 85 % | ~89 % | 200 % | 116 MiB | 512 MiB |
|
|
||||||
| scrabble-tempo | 33 % | — | (none) | **1024 MiB** (at cap) | 1024 MiB |
|
|
||||||
| scrabble-otelcol | 11 % | — | (none) | 131 MiB | 512 MiB |
|
|
||||||
| scrabble-loadtest (harness) | 157 % | — | 300 % | 369 MiB | — |
|
|
||||||
|
|
||||||
- **The gateway is the binding constraint.** With one h2c connection per player it draws
|
|
||||||
~1.45 cores sustained and **bursts to its 2-core cap** at 500 players, throttling
|
|
||||||
briefly — the source of the 2.49 % `transport_error`. R2 saw only ~0.93 core because
|
|
||||||
all 500 players shared one connection; the +~0.5 core is the realistic per-connection
|
|
||||||
overhead (500 separate HTTP/2 connections). This is a sizing fact, not a regression.
|
|
||||||
- **backend is over-provisioned** (~0.85 core vs a 2-core cap); **postgres** (~1.4 cores)
|
|
||||||
has headroom; both stayed ≤ 120 MiB.
|
|
||||||
- **tempo reached its 1 GiB memory cap** (R2: 446 MiB) — an OOM risk under sustained
|
|
||||||
tracing.
|
|
||||||
- **Postgres backends peaked at 28**, with the backend pool at its `MaxOpenConns=25` cap.
|
|
||||||
Cache hit stayed ~100 % (no disk reads); CPU, not I/O, is the limit.
|
|
||||||
- **docker log volume (30 min):** backend 14.2 MiB, gateway 4.6 MiB, postgres 0.04 MiB —
|
|
||||||
the backend's per-request latency line at info dominates, and json-file logs had no
|
|
||||||
rotation.
|
|
||||||
|
|
||||||
## Tuning applied
|
|
||||||
|
|
||||||
Agreed from the profile (all in `deploy/docker-compose.yml`; no code change — the pool
|
|
||||||
is already env-driven):
|
|
||||||
|
|
||||||
| knob | from | to | why |
|
|
||||||
|------|------|----|-----|
|
|
||||||
| gateway CPU + `GOMAXPROCS` | 2 cores / 2 | **3 cores / 3** | it bursts into the 2-core cap at 500 players (the 2.49 % `transport_error`); 3 absorbs the bursts |
|
|
||||||
| tempo memory | 1 GiB | **2 GiB** | it reached the 1 GiB cap (OOM risk) |
|
|
||||||
| backend `MAX_OPEN_CONNS` | 25 | **40** | the pool sat at its 25-conn cap at peak; headroom trims the p99 tail |
|
|
||||||
| docker logs | unbounded | **json-file 10m × 3** | bound the ~14 MiB / 30 min backend log; level stays `info` |
|
|
||||||
|
|
||||||
Left as-is: backend / postgres at 2 cores / 512 MiB (peak ~0.85 / ~1.4 cores — headroom
|
|
||||||
is cheap on the shared host); the per-user rate limiter and `h2cMaxConcurrentStreams=250`
|
|
||||||
(per-connection now, ~1 stream each — ample) and cache TTLs (no pressure observed).
|
|
||||||
|
|
||||||
### Validation re-run
|
|
||||||
|
|
||||||
Re-running the **same gradual ramp** (50 → 200 → 500) on the tuned contour confirms the
|
|
||||||
fix:
|
|
||||||
|
|
||||||
- **`game.state` `transport_error` fell to 0.72 %** (853 / 119 051), down from 2.49 % at
|
|
||||||
2 cores. The latency tail also improved — p99 in the 1000 ms bucket, max 1220 ms (was
|
|
||||||
the 2000 ms bucket, max 2549 ms).
|
|
||||||
- The **gateway peaked at ~2 cores** (≈196 % on the 30 s gauge) — now comfortably **under
|
|
||||||
the 3-core cap**, so it no longer throttles. backend ~1 core, postgres ~1.3 cores.
|
|
||||||
- **tempo peaked at ~1.27 GiB** — under the new 2 GiB cap (it would have OOM-ed at 1 GiB).
|
|
||||||
- Drop-finished still holds (`game_finished` on chat 41/42); the limiter still rejects
|
|
||||||
99.97 % of the hammer at p99 2 ms; `stream errors: 0`.
|
|
||||||
|
|
||||||
A separate **burst stress** (a single 100 → 500 jump — 400 players connecting at once)
|
|
||||||
**pegged the gateway at 3 cores** (≈296 % sustained) and pushed `game.state`
|
|
||||||
`transport_error` to 9.27 %. The gateway is **connection-CPU-bound and bursty**: average
|
|
||||||
load is ~1 core, but a mass-simultaneous connection storm saturates whatever single-node
|
|
||||||
cap it is given. Real arrivals are gradual (the canonical run), where 3 cores has
|
|
||||||
headroom; the lever for a true arrival spike is **horizontal scaling**, not more cores per
|
|
||||||
node — carried into the prod recommendation below.
|
|
||||||
|
|
||||||
## Prod-sizing recommendation (Stage 18)
|
|
||||||
|
|
||||||
The contour is **CPU-bound and gateway-led** at 500 concurrent players. Carry these to the
|
|
||||||
prod contour env (the same compose, `PROD_*` values):
|
|
||||||
|
|
||||||
- **gateway: ≥ 3 cores** per ~500 concurrent players, `GOMAXPROCS` pinned to the limit —
|
|
||||||
it scales with the **connection count**, not just the request rate; beyond one node's
|
|
||||||
worth, scale the gateway **horizontally** rather than vertically.
|
|
||||||
- **backend: ~1–2 cores**, pool 40 — comfortable; the work is light per request.
|
|
||||||
- **postgres: ~2 cores / ≥ 512 MiB** — ~1.4 cores at 500 players, 100 % cache hit.
|
|
||||||
- **tempo: ≥ 2 GiB**; the Go services run under ~170 MiB (256 MiB would suffice, 512 is
|
|
||||||
safe); pin `GOMAXPROCS` to each CPU limit; keep json-file rotation.
|
|
||||||
- Memory is not the constraint anywhere; CPU is.
|
|
||||||
|
|
||||||
### VPS / VDS sizing (single-host contour)
|
|
||||||
|
|
||||||
The whole contour (the app + the observability stack) runs on one host via
|
|
||||||
`docker-compose`. The tiers below are grounded in the R7 profile (**≈5.5 cores / ≈2.5 GiB
|
|
||||||
RAM peak at 500 concurrent players**; ≈0.5 GiB idle) and the **measured** on-disk
|
|
||||||
footprint: prod images ≈2.4 GB; the Tempo volume **3.1 GB at 72 h** retention; Prometheus
|
|
||||||
≈1–2 GB at 15 d; the game DB 23 MiB and growing with history. CPU and disk grow; RAM has
|
|
||||||
the most slack.
|
|
||||||
|
|
||||||
| tier | CPU | RAM | disk | handles |
|
|
||||||
|------|-----|-----|------|---------|
|
|
||||||
| **Minimum** | 2 cores | 2 GiB | 20 GiB | ~up to ~150 concurrent; lower the compose limits (gateway 1.5 / backend·postgres 1 / tempo 1 GiB) to fit the box |
|
|
||||||
| **Average** (reasonable load) | 4 cores | 4 GiB | 40 GiB | ~300–400 concurrent comfortably; the tested 500 with occasional gateway burst-throttling |
|
|
||||||
| **Maximum** (worry-free) | 8 cores | 8 GiB | 80 GiB | 500+ concurrent with full gateway burst headroom (its 3-core cap) + room to grow; the compose limits fit as-is |
|
|
||||||
|
|
||||||
- The per-service limits in `docker-compose.yml` are tuned for the **Average/Maximum**
|
|
||||||
target (the gateway alone caps at 3 cores). On the **Minimum** tier, scale them down to
|
|
||||||
match the host or the caps over-subscribe it.
|
|
||||||
- **Disk is dominated by observability retention + DB growth.** Tempo (72 h traces) and
|
|
||||||
Prometheus (15 d metrics) are the main levers — shorten the windows (or move Tempo to
|
|
||||||
object storage) to cut disk; Postgres grows with game history, so budget for months of
|
|
||||||
it; container logs are already capped (json-file 10m × 3 ≈ 30 MiB each).
|
|
||||||
- **RAM** rarely binds: the contour peaks ≈2.5 GiB at 500 players and the sum of all
|
|
||||||
configured limits is ≈5.6 GiB, so 8 GiB never strains.
|
|
||||||
- Beyond one host's worth of players, scale the **gateway horizontally** (it is
|
|
||||||
connection-CPU-bound) rather than ordering an ever-bigger box.
|
|
||||||
|
|
||||||
## Re-running
|
|
||||||
|
|
||||||
See [`README.md`](README.md). Briefly, from the repo root:
|
|
||||||
|
|
||||||
```sh
|
|
||||||
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
|
||||||
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
|
||||||
-e POSTGRES_PASSWORD=… scrabble-loadtest run --reset --cleanup
|
|
||||||
```
|
|
||||||
|
|
||||||
The harness stays in the repo for future repeats.
|
|
||||||
@@ -0,0 +1,194 @@
|
|||||||
|
# loadtest — stress trip report
|
||||||
|
|
||||||
|
The pre-release stress write-up for [`PRERELEASE.md`](../PRERELEASE.md). It drives the
|
||||||
|
`scrabble/loadtest` harness against a freshly redeployed test contour to confirm the
|
||||||
|
system holds at scale and to settle resource sizing before the prod cutover. The harness
|
||||||
|
stays in the repo for repeats; see [`README.md`](README.md) for how to run it.
|
||||||
|
|
||||||
|
This report supersedes the earlier per-phase notes. The harness has been through three
|
||||||
|
passes: an early diagnostic, a tuning pass that sized container limits / `GOMAXPROCS`, and
|
||||||
|
this final pass — which **added the per-tile `game.evaluate` preview to the model** (the
|
||||||
|
hottest real gameplay call, previously unmodelled) and, with it, surfaced and fixed the
|
||||||
|
**gateway→backend connection-pool bottleneck** described below. The numbers here are from
|
||||||
|
that final pass.
|
||||||
|
|
||||||
|
## What it models
|
||||||
|
|
||||||
|
The harness seeds a large account population with pre-created sessions directly in
|
||||||
|
Postgres, then drives virtual players through the **gateway edge protocol** (h2c) in real
|
||||||
|
games assembled via the invitation flow. Each player owns its own `edge.Client` (its own
|
||||||
|
h2c connection, like a real client), holds a live `Subscribe` stream, and per tick polls
|
||||||
|
`game.state`, replays `game.history`, generates a legal **mid-ranked** move with the
|
||||||
|
embedded `scrabble-solver`, and submits it (or passes/exchanges). A fraction of ticks
|
||||||
|
exercise nudge / chat / check-word / draft / profile / stats. A separate **gateway-hammer**
|
||||||
|
floods `games.list` to verify the rate limiter.
|
||||||
|
|
||||||
|
### The evaluate hot path (this pass)
|
||||||
|
|
||||||
|
A real client previews every tentative play as the user arranges tiles: the UI fires a
|
||||||
|
debounced `game.evaluate` (legality + score) on each placement change while it is the
|
||||||
|
player's turn. Over a single composed word that is **several evaluate calls per turn** —
|
||||||
|
far more than the one `submit_play` — so `game.evaluate` is the single hottest gameplay
|
||||||
|
request at scale. The earlier passes did not model it at all (they submitted directly),
|
||||||
|
which understated the real load.
|
||||||
|
|
||||||
|
This pass models it: when a player composes a play of *K* newly-placed tiles, it fires one
|
||||||
|
`evaluate` per landed tile (a growing prefix of the tiles), plus a small number of
|
||||||
|
full-composition re-previews for reconsideration, spaced by a human-paced gap (the client's
|
||||||
|
250 ms debounce), then one `draft.save`, then `submit_play`. `--eval=false` reproduces the
|
||||||
|
pre-evaluate harness for an A/B baseline; `--eval-recon` tunes the reconsideration count.
|
||||||
|
|
||||||
|
`game.check_word` is a *different*, manual "look this word up" panel (throttled, on demand)
|
||||||
|
— not the per-tile call — and is exercised separately as a secondary op.
|
||||||
|
|
||||||
|
## Final run (eval-on, after the connection-pool fix)
|
||||||
|
|
||||||
|
Contour: backend / postgres capped at 2 cores / 512 MiB (`GOMAXPROCS=2`), gateway at
|
||||||
|
3 cores / 512 MiB (`GOMAXPROCS=3`), per the tuned `deploy/docker-compose.yml`. Gradual ramp
|
||||||
|
**50 → 200 → 500** concurrent players, 4 min/step, `--tick 800ms`, gateway-hammer on. The
|
||||||
|
harness ran as a one-shot container on `scrabble-internal`, capped at `--cpus 3`. The DB was
|
||||||
|
wiped before the run (`DROP SCHEMA backend CASCADE`); the seeded population was removed by
|
||||||
|
`--cleanup` afterwards.
|
||||||
|
|
||||||
|
Per-operation results at the 500-player peak (740 s, gameplay rows; the hammer row is the
|
||||||
|
limiter probe):
|
||||||
|
|
||||||
|
| operation | count | req/s | p50 | p99 | max | notes |
|
||||||
|
|-----------|------:|------:|----:|----:|----:|-------|
|
||||||
|
| game.evaluate | 85 721 | 115.9 | 1 ms | 200 ms | 193 ms | **the hot path** — all ok |
|
||||||
|
| game.state | 115 926 | 156.7 | 100 ms | 200 ms | 260 ms | transport_error 86 (0.07 %) |
|
||||||
|
| game.history | 22 258 | 30.1 | 5 ms | 100 ms | 195 ms | all ok |
|
||||||
|
| draft.save | 23 031 | 31.1 | 2 ms | 200 ms | 194 ms | all ok |
|
||||||
|
| game.submit_play | 21 704 | 29.3 | 1 ms | 200 ms | 274 ms | ok 3 902; not_your_turn / illegal_play are concurrent-play races (see caveat) |
|
||||||
|
| hammer:games.list | 522 756 | 706.7 | 1 ms | 2 ms | 53 ms | **99.97 % rate_limited** — limiter holds |
|
||||||
|
|
||||||
|
- **Volume:** 802 200 total edge calls (1 084 req/s incl. the hammer; ~377 req/s of real
|
||||||
|
gameplay). `stream errors: 0`. Live events: 11 199 `opponent_moved`, 4 153 `your_turn`.
|
||||||
|
- **`game.evaluate` is the dominant gameplay write-path call** at ~116 req/s — second only
|
||||||
|
to the `game.state` poll — and it is cheap: p50 1 ms, effectively zero errors. The backend
|
||||||
|
serves it straight from the in-memory live-game cache; on a warm hit it skips the database
|
||||||
|
entirely (see *Postgres read path* below, which halved its p99 to 100 ms).
|
||||||
|
- **Latency stayed healthy** under the heavier evaluate load: every gameplay op p99 ≤ 200 ms.
|
||||||
|
- **The limiter holds** unchanged: 99.97 % of the hammer rejected at p99 2 ms.
|
||||||
|
|
||||||
|
### Peak CPU (500 players)
|
||||||
|
|
||||||
|
| container | CPU peak | cap |
|
||||||
|
|-----------|---------:|----:|
|
||||||
|
| scrabble-postgres | **165 %** (~1.65 cores) | 200 % |
|
||||||
|
| scrabble-backend | 77 % (~0.77 core) | 200 % |
|
||||||
|
| scrabble-gateway | **26 %** (~0.26 core) | 300 % |
|
||||||
|
| scrabble-loadtest (harness) | 42 % | 300 % |
|
||||||
|
|
||||||
|
Memory stayed modest everywhere (Go services ≤ ~90 MiB). **Postgres is now the busiest
|
||||||
|
service** — it has headroom (1.65 of 2 cores) but is the scaling axis. The gateway, after
|
||||||
|
the fix below, is near-idle.
|
||||||
|
|
||||||
|
## The headline finding: gateway→backend connection churn
|
||||||
|
|
||||||
|
The gateway proxies every synchronous client call to the single backend host over REST.
|
||||||
|
Its backend HTTP client used the default transport, whose **`MaxIdleConnsPerHost` is 2**
|
||||||
|
(`http.DefaultMaxIdleConnsPerHost`). So the gateway kept only **2** keep-alive connections
|
||||||
|
to the backend and opened — then closed — a fresh TCP connection for almost every other
|
||||||
|
call. Measured at the gateway's network namespace:
|
||||||
|
|
||||||
|
| | gateway→backend sockets |
|
||||||
|
|---|---|
|
||||||
|
| before (eval-on, 500 players) | **TIME_WAIT ≈ 26 500**, ESTABLISHED 2 |
|
||||||
|
| after (eval-on, 500 players) | TIME_WAIT ≈ 0 (steady state), **ESTABLISHED ≈ 225 (reused)** |
|
||||||
|
|
||||||
|
26 500 TIME_WAIT sockets is the connection **churn**: ~440 new connections per second,
|
||||||
|
each a full TCP handshake + teardown, the socket then lingering 60 s. That count sits right
|
||||||
|
under the ~28 000 ephemeral-port ceiling — the latent cliff that produced the residual
|
||||||
|
`transport_error` the earlier passes chased on the *client* side (h2c streams) but never
|
||||||
|
eliminated, because the real cause was here, on the *backend* side.
|
||||||
|
|
||||||
|
The fix is one custom `http.Transport` with a wide idle pool
|
||||||
|
(`gateway/internal/backendclient/client.go`, `backendMaxIdleConns`). Before / after, same
|
||||||
|
eval-on workload at 500 players:
|
||||||
|
|
||||||
|
| metric | before fix | after fix |
|
||||||
|
|--------|-----------:|----------:|
|
||||||
|
| gateway→backend TIME_WAIT | ~26 500 | **~0** |
|
||||||
|
| gateway CPU peak | **175 %** (~1.75 cores) | **26 %** (~0.26 core) |
|
||||||
|
| game.state p99 | 500 ms | 200 ms |
|
||||||
|
|
||||||
|
**The churn was burning ~1.5 gateway cores of pure connection setup/teardown.** Removing it
|
||||||
|
cut peak gateway CPU ~7× and erased the port-exhaustion cliff. The backend and postgres CPU
|
||||||
|
are unchanged — they do the real work; only the gateway's wasted overhead disappeared. The
|
||||||
|
pool settles at ~225 live connections at 500 players; the constant is set to 512 for ~2×
|
||||||
|
headroom.
|
||||||
|
|
||||||
|
## Sizing — why the old "≈150 concurrent / 2-core" figure was a bug, not a floor
|
||||||
|
|
||||||
|
The earlier tuning pass concluded the gateway was the binding constraint — "size it for
|
||||||
|
≥ 3 cores per 500 players, scale it horizontally" — and the single-host "minimum" tier
|
||||||
|
topped out near ~150 concurrent. **That was sizing around the connection-churn bug.** The
|
||||||
|
gateway drew ~1.75–3 cores not from proxying work but from churning backend connections;
|
||||||
|
the backend behind it sat near-idle the whole time.
|
||||||
|
|
||||||
|
With the churn fixed, at **500 concurrent players** the app draws roughly:
|
||||||
|
|
||||||
|
- **gateway ≈ 0.26 core** (was ~3) — no longer the constraint,
|
||||||
|
- **backend ≈ 0.77 core**,
|
||||||
|
- **postgres ≈ 1.65 cores** — now the busiest, with headroom,
|
||||||
|
|
||||||
|
≈ **2.7 app cores total** (down from the ~5.5-core contour peak the tuning pass recorded,
|
||||||
|
*and* under a heavier, more realistic workload that now includes `game.evaluate`). Postgres,
|
||||||
|
not the gateway, is the scaling axis.
|
||||||
|
|
||||||
|
Revised single-host guidance (app + co-resident observability stack on one box):
|
||||||
|
|
||||||
|
| tier | CPU | RAM | handles |
|
||||||
|
|------|-----|-----|---------|
|
||||||
|
| **Minimum** | 2 cores | 2 GiB | comfortably the low hundreds of concurrent — the gateway no longer eats cores; postgres + the observability stack set the limit |
|
||||||
|
| **Average** | 4 cores | 4 GiB | 500 concurrent with headroom |
|
||||||
|
| **Maximum** | 8 cores | 8 GiB | 500+ with full burst headroom and room to grow |
|
||||||
|
|
||||||
|
The gateway's compose limit can drop well below its old 3 cores; it is now connection-pool
|
||||||
|
bound, not connection-CPU bound. Memory was never the constraint. Disk is still dominated
|
||||||
|
by observability retention (Tempo, Prometheus) + DB growth — unchanged from before.
|
||||||
|
|
||||||
|
## Postgres read path (warm-cache optimization)
|
||||||
|
|
||||||
|
Following this pass, `game.evaluate` no longer reads the database on the hot path. An
|
||||||
|
active game is already resident in the in-memory live-game cache (mutated in place across
|
||||||
|
moves, evicted only on finish), so the preview answers its seat-membership check from the
|
||||||
|
cached immutable seat list and scores against the cached engine game — **no `GetGame` on a
|
||||||
|
warm hit**. `GetGame` itself was also folded from two round-trips (game, then seats) into a
|
||||||
|
single `LEFT JOIN`. Measured at 500 players, **`game.evaluate` p99 halved (200 → 100 ms)**
|
||||||
|
and the per-operation query count dropped.
|
||||||
|
|
||||||
|
It did **not** cut postgres CPU, and the measurement says why: postgres is **write-bound**,
|
||||||
|
not read-bound. `pg_stat_user_tables` puts the cost in the per-move `CommitMove`
|
||||||
|
transaction (a `game_moves` insert plus `games` / `game_players` updates), the debounced
|
||||||
|
`game_drafts` upserts (~60 k in one run), and the journal replays — not the cheap, indexed,
|
||||||
|
fully-cached `GetGame` lookups this change removed (one re-run even committed 28 % more
|
||||||
|
plays, whose extra writes masked the saved reads). Postgres also runs with headroom
|
||||||
|
(~1.5 of 2 cores), and the gateway fix freed ~3 cores on the box, so the lever if postgres
|
||||||
|
ever caps is **more cores** (it is CPU-bound, not I/O), not riskier write-path surgery. So
|
||||||
|
this change is a latency / query-volume win, deliberately not a DB-CPU one.
|
||||||
|
|
||||||
|
## Caveat — harness fidelity
|
||||||
|
|
||||||
|
The harness's `not_your_turn` and `illegal_play` on `submit_play` are concurrent-play
|
||||||
|
artifacts, not system errors: it generates a move from a locally replayed board, and a
|
||||||
|
fast opponent (or a transport hiccup) can move between the state fetch and the submit,
|
||||||
|
leaving the move out of turn or illegal on the now-changed board. A real client previews
|
||||||
|
with `evaluate` and only submits a legal, in-turn play. These rejections are cheap domain
|
||||||
|
outcomes (HTTP-ok with a stable code) and do not change the request *load*, which is what
|
||||||
|
the run measures. The harness also shares the host CPU with the contour (capped with
|
||||||
|
`--cpus`); a fully isolated ceiling on separate hardware remains future work.
|
||||||
|
|
||||||
|
## Re-running
|
||||||
|
|
||||||
|
From the repo root:
|
||||||
|
|
||||||
|
```sh
|
||||||
|
docker build -f loadtest/Dockerfile -t scrabble-loadtest .
|
||||||
|
docker run --rm --cpus=3 --name scrabble-loadtest --network scrabble-internal \
|
||||||
|
-e POSTGRES_PASSWORD="$TEST_POSTGRES_PASSWORD" scrabble-loadtest run --reset --cleanup
|
||||||
|
```
|
||||||
|
|
||||||
|
`--eval=false` reproduces the pre-evaluate baseline for comparison. The authoritative hard
|
||||||
|
reset of the contour DB remains `DROP SCHEMA backend CASCADE` + a backend restart.
|
||||||
@@ -73,6 +73,8 @@ func cmdRun(ctx context.Context, log *slog.Logger, args []string) error {
|
|||||||
gpp := fs.Int("games-per-player", 0, "target concurrent games per player (0 => random 3..5)")
|
gpp := fs.Int("games-per-player", 0, "target concurrent games per player (0 => random 3..5)")
|
||||||
tick := fs.Duration("tick", 800*time.Millisecond, "per-player operation cadence")
|
tick := fs.Duration("tick", 800*time.Millisecond, "per-player operation cadence")
|
||||||
secProb := fs.Float64("secondary-prob", 0.08, "chance per tick of a non-move operation")
|
secProb := fs.Float64("secondary-prob", 0.08, "chance per tick of a non-move operation")
|
||||||
|
eval := fs.Bool("eval", true, "model the per-tile evaluate preview (the realistic gameplay hot path); --eval=false reproduces the pre-evaluate harness for an A/B baseline")
|
||||||
|
evalRecon := fs.Int("eval-recon", 1, "extra full-composition evaluate re-previews per play (reconsideration), beyond one per placed tile")
|
||||||
hammerWorkers := fs.Int("hammer-workers", 20, "gateway-hammer concurrent callers (0 disables)")
|
hammerWorkers := fs.Int("hammer-workers", 20, "gateway-hammer concurrent callers (0 disables)")
|
||||||
hammerDur := fs.Duration("hammer-dur", 15*time.Second, "gateway-hammer duration")
|
hammerDur := fs.Duration("hammer-dur", 15*time.Second, "gateway-hammer duration")
|
||||||
reset := fs.Bool("reset", false, "delete prior harness rows before seeding")
|
reset := fs.Bool("reset", false, "delete prior harness rows before seeding")
|
||||||
@@ -117,6 +119,7 @@ func cmdRun(ctx context.Context, log *slog.Logger, args []string) error {
|
|||||||
cfg := scenario.RealisticConfig{
|
cfg := scenario.RealisticConfig{
|
||||||
Steps: steps, StepDur: *stepDur, GamesPerPlayer: *gpp,
|
Steps: steps, StepDur: *stepDur, GamesPerPlayer: *gpp,
|
||||||
Tick: *tick, SecondaryProb: *secProb,
|
Tick: *tick, SecondaryProb: *secProb,
|
||||||
|
Eval: *eval, EvalRecon: *evalRecon,
|
||||||
}
|
}
|
||||||
if err := drv.RunRealistic(ctx, pool, cfg); err != nil && !errors.Is(err, context.Canceled) {
|
if err := drv.RunRealistic(ctx, pool, cfg); err != nil && !errors.Is(err, context.Canceled) {
|
||||||
return err
|
return err
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ const (
|
|||||||
msgSubmitPlay = "game.submit_play"
|
msgSubmitPlay = "game.submit_play"
|
||||||
msgPass = "game.pass"
|
msgPass = "game.pass"
|
||||||
msgExchange = "game.exchange"
|
msgExchange = "game.exchange"
|
||||||
|
msgEvaluate = "game.evaluate"
|
||||||
msgState = "game.state"
|
msgState = "game.state"
|
||||||
msgHistory = "game.history"
|
msgHistory = "game.history"
|
||||||
msgGamesList = "games.list"
|
msgGamesList = "games.list"
|
||||||
|
|||||||
@@ -63,6 +63,33 @@ func submitPlay(gameID string, tiles []PlayTile) []byte {
|
|||||||
return b.FinishedBytes()
|
return b.FinishedBytes()
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// evalReq builds an EvalRequest payload (game id plus the tentative newly-placed tiles).
|
||||||
|
// It mirrors submitPlay's shape — the backend infers the play's orientation the same way —
|
||||||
|
// so a preview previews exactly what submitting those tiles would score.
|
||||||
|
func evalReq(gameID string, tiles []PlayTile) []byte {
|
||||||
|
b := flatbuffers.NewBuilder(256)
|
||||||
|
gid := b.CreateString(gameID)
|
||||||
|
offs := make([]flatbuffers.UOffsetT, len(tiles))
|
||||||
|
for i, t := range tiles {
|
||||||
|
fb.PlayTileStart(b)
|
||||||
|
fb.PlayTileAddRow(b, int32(t.Row))
|
||||||
|
fb.PlayTileAddCol(b, int32(t.Col))
|
||||||
|
fb.PlayTileAddLetter(b, t.Letter)
|
||||||
|
fb.PlayTileAddBlank(b, t.Blank)
|
||||||
|
offs[i] = fb.PlayTileEnd(b)
|
||||||
|
}
|
||||||
|
fb.EvalRequestStartTilesVector(b, len(offs))
|
||||||
|
for i := len(offs) - 1; i >= 0; i-- {
|
||||||
|
b.PrependUOffsetT(offs[i])
|
||||||
|
}
|
||||||
|
tilesVec := b.EndVector(len(offs))
|
||||||
|
fb.EvalRequestStart(b)
|
||||||
|
fb.EvalRequestAddGameId(b, gid)
|
||||||
|
fb.EvalRequestAddTiles(b, tilesVec)
|
||||||
|
b.Finish(fb.EvalRequestEnd(b))
|
||||||
|
return b.FinishedBytes()
|
||||||
|
}
|
||||||
|
|
||||||
// exchange builds an ExchangeRequest payload swapping the listed rack tiles (alphabet
|
// exchange builds an ExchangeRequest payload swapping the listed rack tiles (alphabet
|
||||||
// indices; 255 a blank).
|
// indices; 255 a blank).
|
||||||
func exchange(gameID string, tiles []byte) []byte {
|
func exchange(gameID string, tiles []byte) []byte {
|
||||||
|
|||||||
@@ -53,6 +53,15 @@ func (c *Client) Exchange(ctx context.Context, token, gameID string, tiles []byt
|
|||||||
return decodeMoveResultGame(r.Payload), r.Code, nil
|
return decodeMoveResultGame(r.Payload), r.Code, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Evaluate previews a tentative play's legality and score without committing it. It is
|
||||||
|
// the per-tile composition call a real client fires (debounced) on every change while
|
||||||
|
// arranging a word, so it is the hottest gameplay request at scale. The harness records
|
||||||
|
// only the result code and latency; an illegal preview is a successful "ok" call.
|
||||||
|
func (c *Client) Evaluate(ctx context.Context, token, gameID string, tiles []PlayTile) (string, error) {
|
||||||
|
r, err := c.execute(ctx, token, msgEvaluate, evalReq(gameID, tiles))
|
||||||
|
return r.Code, err
|
||||||
|
}
|
||||||
|
|
||||||
// Nudge prods the opponent whose turn it is.
|
// Nudge prods the opponent whose turn it is.
|
||||||
func (c *Client) Nudge(ctx context.Context, token, gameID string) (string, error) {
|
func (c *Client) Nudge(ctx context.Context, token, gameID string) (string, error) {
|
||||||
r, err := c.execute(ctx, token, msgNudge, gameAction(gameID))
|
r, err := c.execute(ctx, token, msgNudge, gameAction(gameID))
|
||||||
|
|||||||
@@ -42,19 +42,35 @@ type RealisticConfig struct {
|
|||||||
GamesPerPlayer int // target concurrent games per player; 0 => random 3..5
|
GamesPerPlayer int // target concurrent games per player; 0 => random 3..5
|
||||||
Tick time.Duration // per-player operation cadence (keeps a player under the per-user limit)
|
Tick time.Duration // per-player operation cadence (keeps a player under the per-user limit)
|
||||||
SecondaryProb float64 // chance per tick of a non-move operation
|
SecondaryProb float64 // chance per tick of a non-move operation
|
||||||
|
Eval bool // model the per-tile evaluate preview (the gameplay hot path); false reproduces the pre-evaluate harness
|
||||||
|
EvalRecon int // extra full-composition evaluate re-previews per play, beyond one per placed tile
|
||||||
}
|
}
|
||||||
|
|
||||||
// DefaultRealistic returns the moderate ramp: 50 -> 200
|
// DefaultRealistic returns the moderate ramp: 50 -> 200
|
||||||
// -> 500 concurrent players, ~12 minutes per step, ~1 op/s per player.
|
// -> 500 concurrent players, ~12 minutes per step, ~1 op/s per player, with the
|
||||||
|
// per-tile evaluate preview modelled (the realistic hot path).
|
||||||
func DefaultRealistic() RealisticConfig {
|
func DefaultRealistic() RealisticConfig {
|
||||||
return RealisticConfig{
|
return RealisticConfig{
|
||||||
Steps: []int{50, 200, 500},
|
Steps: []int{50, 200, 500},
|
||||||
StepDur: 12 * time.Minute,
|
StepDur: 12 * time.Minute,
|
||||||
Tick: 800 * time.Millisecond,
|
Tick: 800 * time.Millisecond,
|
||||||
SecondaryProb: 0.08,
|
SecondaryProb: 0.08,
|
||||||
|
Eval: true,
|
||||||
|
EvalRecon: 1,
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// evalGapBase and evalGapSpan bound the modelled pause between successive tile
|
||||||
|
// placements: the client's 250 ms debounce coalesces faster drags into a single
|
||||||
|
// evaluate, so a thoughtful player's previews are spaced by a gap drawn from
|
||||||
|
// [base, base+span] — wide enough that a normal composition stays under the per-user
|
||||||
|
// rate limit, the way a real one does (the limiter's cost is measured by the hammer,
|
||||||
|
// not by self-inflicted rejections here).
|
||||||
|
const (
|
||||||
|
evalGapBase = 250 * time.Millisecond
|
||||||
|
evalGapSpan = 500 * time.Millisecond
|
||||||
|
)
|
||||||
|
|
||||||
// RunRealistic runs the staged ramp. Each step activates more players (drawn from the
|
// RunRealistic runs the staged ramp. Each step activates more players (drawn from the
|
||||||
// seeded pool), assembles a cohort of games for them and starts their turn loops; the
|
// seeded pool), assembles a cohort of games for them and starts their turn loops; the
|
||||||
// loops run until the whole ramp ends. Players from earlier steps keep playing, so
|
// loops run until the whole ramp ends. Players from earlier steps keep playing, so
|
||||||
@@ -128,7 +144,7 @@ func (d *Driver) playerLoop(ctx context.Context, p seed.Account, games []*Game,
|
|||||||
d.secondaryOp(ctx, c, p, g, rng)
|
d.secondaryOp(ctx, c, p, g, rng)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
if d.playTurn(ctx, c, p, g, rng) {
|
if d.playTurn(ctx, c, p, g, cfg, rng) {
|
||||||
active = slices.DeleteFunc(active, func(x *Game) bool { return x == g })
|
active = slices.DeleteFunc(active, func(x *Game) bool { return x == g })
|
||||||
gi = 0
|
gi = 0
|
||||||
if len(active) == 0 {
|
if len(active) == 0 {
|
||||||
@@ -161,10 +177,10 @@ func (d *Driver) subscribeLoop(ctx context.Context, c *edge.Client, p seed.Accou
|
|||||||
}
|
}
|
||||||
|
|
||||||
// playTurn plays one turn in g over the player's client when it is the player's
|
// playTurn plays one turn in g over the player's client when it is the player's
|
||||||
// move: fetch state, replay history, pick a legal move and submit it (or exchange /
|
// move: fetch state, replay history, pick a legal move, compose it (the per-tile
|
||||||
// pass). It reports whether the game has finished, so the caller can drop it from the
|
// evaluate previews a real client fires) and submit it (or exchange / pass). It reports
|
||||||
// rotation.
|
// whether the game has finished, so the caller can drop it from the rotation.
|
||||||
func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g *Game, rng *rand.Rand) (finished bool) {
|
func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g *Game, cfg RealisticConfig, rng *rand.Rand) (finished bool) {
|
||||||
seat := g.seatOf(p.ID.String())
|
seat := g.seatOf(p.ID.String())
|
||||||
if seat < 0 {
|
if seat < 0 {
|
||||||
return false
|
return false
|
||||||
@@ -196,6 +212,7 @@ func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g
|
|||||||
}
|
}
|
||||||
switch action.Kind {
|
switch action.Kind {
|
||||||
case "play":
|
case "play":
|
||||||
|
d.composePlay(ctx, c, p, g, action.Tiles, cfg, rng)
|
||||||
t0 = time.Now()
|
t0 = time.Now()
|
||||||
_, code, _ := c.SubmitPlay(ctx, p.Token, g.ID, action.Tiles)
|
_, code, _ := c.SubmitPlay(ctx, p.Token, g.ID, action.Tiles)
|
||||||
d.rec.Record("game.submit_play", code, time.Since(t0))
|
d.rec.Record("game.submit_play", code, time.Since(t0))
|
||||||
@@ -211,6 +228,59 @@ func (d *Driver) playTurn(ctx context.Context, c *edge.Client, p seed.Account, g
|
|||||||
return false
|
return false
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// composePlay models a player arranging the chosen play tile by tile before committing:
|
||||||
|
// the debounced evaluate preview the real client fires on each placement (a growing prefix
|
||||||
|
// of the tiles), a few full-composition re-previews for reconsideration (recall a tile, try
|
||||||
|
// another spot), and the single draft persistence the client debounces out. evaluate is the
|
||||||
|
// hottest gameplay request at scale, so omitting it (the pre-evaluate harness) understated
|
||||||
|
// the load; cfg.Eval false reproduces that baseline for an A/B comparison. Every step
|
||||||
|
// honours ctx, so end-of-run cancellation never blocks on a sleep or an in-flight preview.
|
||||||
|
func (d *Driver) composePlay(ctx context.Context, c *edge.Client, p seed.Account, g *Game, tiles []edge.PlayTile, cfg RealisticConfig, rng *rand.Rand) {
|
||||||
|
if !cfg.Eval || len(tiles) == 0 {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
// One evaluate per landed tile: the growing prefix mirrors the client re-previewing
|
||||||
|
// after each placement (an early prefix is often illegal, which is still a successful
|
||||||
|
// "ok" round trip — exactly the backend work a real composition triggers).
|
||||||
|
for n := 1; n <= len(tiles); n++ {
|
||||||
|
if !jitterSleep(ctx, rng, evalGapBase, evalGapSpan) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t0 := time.Now()
|
||||||
|
code, _ := c.Evaluate(ctx, p.Token, g.ID, tiles[:n])
|
||||||
|
d.rec.Record("game.evaluate", code, time.Since(t0))
|
||||||
|
}
|
||||||
|
for r := 0; r < cfg.EvalRecon; r++ {
|
||||||
|
if !jitterSleep(ctx, rng, evalGapBase, evalGapSpan) {
|
||||||
|
return
|
||||||
|
}
|
||||||
|
t0 := time.Now()
|
||||||
|
code, _ := c.Evaluate(ctx, p.Token, g.ID, tiles)
|
||||||
|
d.rec.Record("game.evaluate", code, time.Since(t0))
|
||||||
|
}
|
||||||
|
// The client persists the in-progress composition (debounced to one upsert). Its opaque
|
||||||
|
// JSON content does not affect the call's cost, so a minimal valid shape stands in.
|
||||||
|
t0 := time.Now()
|
||||||
|
code, _ := c.DraftSave(ctx, p.Token, g.ID, `{"rack_order":"","board_tiles":[]}`)
|
||||||
|
d.rec.Record("draft.save", code, time.Since(t0))
|
||||||
|
}
|
||||||
|
|
||||||
|
// jitterSleep pauses for a randomised gap in [base, base+span], modelling the human pause
|
||||||
|
// between tile placements that the client's debounce coalesces into one evaluate. It
|
||||||
|
// returns false if ctx is cancelled during the wait, so a composition unwinds promptly at
|
||||||
|
// end of run.
|
||||||
|
func jitterSleep(ctx context.Context, rng *rand.Rand, base, span time.Duration) bool {
|
||||||
|
d := base + time.Duration(rng.Int63n(int64(span)+1))
|
||||||
|
t := time.NewTimer(d)
|
||||||
|
defer t.Stop()
|
||||||
|
select {
|
||||||
|
case <-ctx.Done():
|
||||||
|
return false
|
||||||
|
case <-t.C:
|
||||||
|
return true
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
// secondaryOp exercises one of the non-move edge operations the plan calls out, so
|
// secondaryOp exercises one of the non-move edge operations the plan calls out, so
|
||||||
// the run touches nudge / chat / check-word / draft / profile / stats too, over the
|
// the run touches nudge / chat / check-word / draft / profile / stats too, over the
|
||||||
// player's own client.
|
// player's own client.
|
||||||
|
|||||||
@@ -29,6 +29,8 @@ import (
|
|||||||
"go.opentelemetry.io/otel/sdk/resource"
|
"go.opentelemetry.io/otel/sdk/resource"
|
||||||
sdktrace "go.opentelemetry.io/otel/sdk/trace"
|
sdktrace "go.opentelemetry.io/otel/sdk/trace"
|
||||||
"go.opentelemetry.io/otel/trace"
|
"go.opentelemetry.io/otel/trace"
|
||||||
|
|
||||||
|
"scrabble/pkg/version"
|
||||||
)
|
)
|
||||||
|
|
||||||
// Exporter selectors supported per signal.
|
// Exporter selectors supported per signal.
|
||||||
@@ -95,9 +97,7 @@ func New(ctx context.Context, cfg Config) (*Runtime, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
res, err := resource.New(ctx, resource.WithAttributes(
|
res, err := serviceResource(ctx, cfg)
|
||||||
attribute.String("service.name", cfg.ServiceName),
|
|
||||||
))
|
|
||||||
if err != nil {
|
if err != nil {
|
||||||
return nil, fmt.Errorf("telemetry: build resource: %w", err)
|
return nil, fmt.Errorf("telemetry: build resource: %w", err)
|
||||||
}
|
}
|
||||||
@@ -122,6 +122,16 @@ func New(ctx context.Context, cfg Config) (*Runtime, error) {
|
|||||||
return &Runtime{tracerProvider: tracerProvider, meterProvider: meterProvider}, nil
|
return &Runtime{tracerProvider: tracerProvider, meterProvider: meterProvider}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// serviceResource builds the OpenTelemetry resource describing this service: its
|
||||||
|
// service.name and the service.version stamped into the binary at build time
|
||||||
|
// (pkg/version, set from the git tag by the deploy).
|
||||||
|
func serviceResource(ctx context.Context, cfg Config) (*resource.Resource, error) {
|
||||||
|
return resource.New(ctx, resource.WithAttributes(
|
||||||
|
attribute.String("service.name", cfg.ServiceName),
|
||||||
|
attribute.String("service.version", version.Version),
|
||||||
|
))
|
||||||
|
}
|
||||||
|
|
||||||
// TracerProvider returns the runtime tracer provider, or the global one when r is
|
// TracerProvider returns the runtime tracer provider, or the global one when r is
|
||||||
// not initialised.
|
// not initialised.
|
||||||
func (r *Runtime) TracerProvider() trace.TracerProvider {
|
func (r *Runtime) TracerProvider() trace.TracerProvider {
|
||||||
|
|||||||
@@ -4,6 +4,8 @@ import (
|
|||||||
"context"
|
"context"
|
||||||
"testing"
|
"testing"
|
||||||
"time"
|
"time"
|
||||||
|
|
||||||
|
"scrabble/pkg/version"
|
||||||
)
|
)
|
||||||
|
|
||||||
// TestConfigValidate covers the supported and rejected exporter selections.
|
// TestConfigValidate covers the supported and rejected exporter selections.
|
||||||
@@ -82,3 +84,22 @@ func TestNilRuntime(t *testing.T) {
|
|||||||
t.Errorf("nil runtime Shutdown: %v", err)
|
t.Errorf("nil runtime Shutdown: %v", err)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// TestServiceResource checks the resource carries service.name and the embedded
|
||||||
|
// service.version (pkg/version, stamped at build time).
|
||||||
|
func TestServiceResource(t *testing.T) {
|
||||||
|
res, err := serviceResource(context.Background(), DefaultConfig("svc"))
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("serviceResource: %v", err)
|
||||||
|
}
|
||||||
|
attrs := map[string]string{}
|
||||||
|
for _, kv := range res.Attributes() {
|
||||||
|
attrs[string(kv.Key)] = kv.Value.AsString()
|
||||||
|
}
|
||||||
|
if attrs["service.name"] != "svc" {
|
||||||
|
t.Errorf("service.name = %q, want svc", attrs["service.name"])
|
||||||
|
}
|
||||||
|
if attrs["service.version"] != version.Version {
|
||||||
|
t.Errorf("service.version = %q, want %q", attrs["service.version"], version.Version)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,10 @@
|
|||||||
|
// Package version exposes the build version stamped into every Scrabble service
|
||||||
|
// binary. The default is "dev"; release builds override it through the linker
|
||||||
|
// (`go build -ldflags "-X scrabble/pkg/version.Version=<value>"`), wired from the
|
||||||
|
// VERSION build-arg in each service Dockerfile, which the deploy sets to the git
|
||||||
|
// tag (`git describe --tags`). It surfaces as the OpenTelemetry service.version
|
||||||
|
// resource attribute (see pkg/telemetry) and the SPA About screen.
|
||||||
|
package version
|
||||||
|
|
||||||
|
// Version is the build version, "dev" unless overridden at link time.
|
||||||
|
var Version = "dev"
|
||||||
@@ -19,8 +19,10 @@ COPY platform/telegram ./platform/telegram
|
|||||||
# Reduce the workspace to what the platform needs: only pkg + platform/telegram.
|
# Reduce the workspace to what the platform needs: only pkg + platform/telegram.
|
||||||
RUN go work edit -dropuse=./backend -dropuse=./gateway -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0 -dropreplace=scrabble-solver
|
RUN go work edit -dropuse=./backend -dropuse=./gateway -dropuse=./loadtest -dropreplace=scrabble/gateway@v0.0.0 -dropreplace=scrabble-solver
|
||||||
|
|
||||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/validator ./platform/telegram/cmd/validator
|
# VERSION (the deploy passes the git tag) is stamped into both binaries via the linker.
|
||||||
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -o /out/bot ./platform/telegram/cmd/bot
|
ARG VERSION=dev
|
||||||
|
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/validator ./platform/telegram/cmd/validator
|
||||||
|
RUN CGO_ENABLED=0 GOOS=linux go build -trimpath -ldflags "-X scrabble/pkg/version.Version=${VERSION}" -o /out/bot ./platform/telegram/cmd/bot
|
||||||
|
|
||||||
# --- validator (home) --------------------------------------------------------
|
# --- validator (home) --------------------------------------------------------
|
||||||
FROM gcr.io/distroless/static-debian12:nonroot AS validator
|
FROM gcr.io/distroless/static-debian12:nonroot AS validator
|
||||||
|
|||||||
@@ -191,6 +191,13 @@ func (t *Bot) handleStart(ctx context.Context, api *tgbot.Bot, update *models.Up
|
|||||||
if update.Message == nil {
|
if update.Message == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
// Reply only in a private chat: the Mini App launch button is an inline web_app
|
||||||
|
// button, which Telegram permits only in private chats — replying to a group message
|
||||||
|
// (the bot is an admin in the moderated chat and now receives its messages) fails with
|
||||||
|
// BUTTON_TYPE_INVALID. In the group the bot only manages permissions, it never chats.
|
||||||
|
if update.Message.Chat.Type != models.ChatTypePrivate {
|
||||||
|
return
|
||||||
|
}
|
||||||
startParam := startPayload(update.Message.Text)
|
startParam := startPayload(update.Message.Text)
|
||||||
if _, err := api.SendMessage(ctx, &tgbot.SendMessageParams{
|
if _, err := api.SendMessage(ctx, &tgbot.SendMessageParams{
|
||||||
ChatID: update.Message.Chat.ID,
|
ChatID: update.Message.Chat.ID,
|
||||||
|
|||||||
@@ -8,6 +8,7 @@ import (
|
|||||||
"strings"
|
"strings"
|
||||||
"testing"
|
"testing"
|
||||||
|
|
||||||
|
"github.com/go-telegram/bot/models"
|
||||||
"go.uber.org/zap"
|
"go.uber.org/zap"
|
||||||
)
|
)
|
||||||
|
|
||||||
@@ -103,6 +104,29 @@ func TestTestEnvironmentRoutesGetMe(t *testing.T) {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestHandleStartRepliesPrivateOnly(t *testing.T) {
|
||||||
|
t.Run("private replies", func(t *testing.T) {
|
||||||
|
api := &fakeBotAPI{}
|
||||||
|
b := newTestBot(t, api)
|
||||||
|
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||||
|
Chat: models.Chat{ID: 42, Type: models.ChatTypePrivate}, Text: "/start g7",
|
||||||
|
}})
|
||||||
|
if api.chatID != "42" || !strings.Contains(api.replyMarkup, "web_app") {
|
||||||
|
t.Errorf("private /start: chat=%q markup=%q, want a web_app reply", api.chatID, api.replyMarkup)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
t.Run("group ignored", func(t *testing.T) {
|
||||||
|
api := &fakeBotAPI{}
|
||||||
|
b := newTestBot(t, api)
|
||||||
|
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||||
|
Chat: models.Chat{ID: -100, Type: models.ChatTypeSupergroup}, Text: "/start",
|
||||||
|
}})
|
||||||
|
if api.chatID != "" {
|
||||||
|
t.Errorf("group /start got a reply (chat=%q); an inline web_app button is invalid in groups", api.chatID)
|
||||||
|
}
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
func TestStartPayload(t *testing.T) {
|
func TestStartPayload(t *testing.T) {
|
||||||
cases := map[string]string{
|
cases := map[string]string{
|
||||||
"/start g123": "g123",
|
"/start g123": "g123",
|
||||||
|
|||||||
@@ -92,6 +92,11 @@ func (t *Bot) handleStart(ctx context.Context, api *tgbot.Bot, update *models.Up
|
|||||||
if update.Message == nil {
|
if update.Message == nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
// Only respond to a private /start: the promo bot is a one-on-one onboarding entry
|
||||||
|
// point and should never reply to group messages.
|
||||||
|
if update.Message.Chat.Type != models.ChatTypePrivate {
|
||||||
|
return
|
||||||
|
}
|
||||||
if err := t.throttle(ctx); err != nil {
|
if err := t.throttle(ctx); err != nil {
|
||||||
return
|
return
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -85,7 +85,7 @@ func TestHandleStartReplies(t *testing.T) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||||
Chat: models.Chat{ID: 42},
|
Chat: models.Chat{ID: 42, Type: models.ChatTypePrivate},
|
||||||
From: &models.User{LanguageCode: "ru"},
|
From: &models.User{LanguageCode: "ru"},
|
||||||
Text: "/start f99",
|
Text: "/start f99",
|
||||||
}})
|
}})
|
||||||
@@ -103,3 +103,19 @@ func TestHandleStartReplies(t *testing.T) {
|
|||||||
t.Errorf("reply_markup = %q, want startapp=f99 (the /start payload forwarded)", api.replyMarkup)
|
t.Errorf("reply_markup = %q, want startapp=f99 (the /start payload forwarded)", api.replyMarkup)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestHandleStartIgnoresGroup(t *testing.T) {
|
||||||
|
api := &fakeAPI{}
|
||||||
|
srv := httptest.NewServer(api)
|
||||||
|
t.Cleanup(srv.Close)
|
||||||
|
b, err := New(Config{Token: "1:2", APIBaseURL: srv.URL, BotUsername: "B", BotLinkURL: "https://t.me/b/a"}, zap.NewNop())
|
||||||
|
if err != nil {
|
||||||
|
t.Fatalf("new: %v", err)
|
||||||
|
}
|
||||||
|
b.handleStart(context.Background(), b.api, &models.Update{Message: &models.Message{
|
||||||
|
Chat: models.Chat{ID: -100, Type: models.ChatTypeSupergroup}, Text: "/start",
|
||||||
|
}})
|
||||||
|
if api.chatID != "" {
|
||||||
|
t.Errorf("replied to a group message (chat=%q); want none", api.chatID)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -238,7 +238,10 @@ test('profile edit disables Save and flags an invalid display name', async ({ pa
|
|||||||
await expect(save).toBeEnabled();
|
await expect(save).toBeEnabled();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('link account: a taken email opens the irreversible merge confirmation', async ({ page }) => {
|
// Account linking is hidden in Profile.svelte while we target provider sign-in (the anonymous
|
||||||
|
// /app/ guest who upgrades by linking comes later). The flow is kept wired; re-enable these two
|
||||||
|
// specs together with the `.emailbox` section.
|
||||||
|
test.skip('link account: a taken email opens the irreversible merge confirmation', async ({ page }) => {
|
||||||
await loginLobby(page);
|
await loginLobby(page);
|
||||||
await openProfile(page);
|
await openProfile(page);
|
||||||
|
|
||||||
@@ -258,7 +261,7 @@ test('link account: a taken email opens the irreversible merge confirmation', as
|
|||||||
await expect(page.getByText('Merge accounts?')).toBeHidden();
|
await expect(page.getByText('Merge accounts?')).toBeHidden();
|
||||||
});
|
});
|
||||||
|
|
||||||
test('link account: the Telegram web sign-in control is offered in a browser', async ({ page }) => {
|
test.skip('link account: the Telegram web sign-in control is offered in a browser', async ({ page }) => {
|
||||||
await loginLobby(page);
|
await loginLobby(page);
|
||||||
await openProfile(page);
|
await openProfile(page);
|
||||||
await expect(page.getByRole('button', { name: 'Link Telegram' })).toBeVisible();
|
await expect(page.getByRole('button', { name: 'Link Telegram' })).toBeVisible();
|
||||||
|
|||||||
@@ -83,6 +83,30 @@ test('tg-fullscreen header keeps a constant native-nav gap as the font scales',
|
|||||||
expect(large.overflows).toBe(false);
|
expect(large.overflows).toBe(false);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('inside Telegram, a failed launch shows the retry screen, not the web login', async ({ page }) => {
|
||||||
|
// initData carrying the mock's "bootfail" sentinel makes authTelegram reject, simulating a
|
||||||
|
// backend outage during launch (e.g. a deploy rolling). The Mini App must surface its own
|
||||||
|
// boot-error/retry screen and never fall back to the web (guest/email) login.
|
||||||
|
await page.addInitScript(() => {
|
||||||
|
Object.assign(window, {
|
||||||
|
Telegram: {
|
||||||
|
WebApp: {
|
||||||
|
initData: 'query_id=bootfail&user=%7B%22id%22%3A1%7D&auth_date=1&hash=deadbeef',
|
||||||
|
initDataUnsafe: {},
|
||||||
|
ready() {},
|
||||||
|
expand() {},
|
||||||
|
},
|
||||||
|
},
|
||||||
|
});
|
||||||
|
});
|
||||||
|
await page.goto('/');
|
||||||
|
|
||||||
|
// After the silent retries, the boot-error screen with its Retry button shows…
|
||||||
|
await expect(page.getByRole('button', { name: 'Retry' })).toBeVisible();
|
||||||
|
// …and the web login (guest) is never shown inside Telegram.
|
||||||
|
await expect(page.getByRole('button', { name: /guest/i })).toHaveCount(0);
|
||||||
|
});
|
||||||
|
|
||||||
test('outside Telegram, the /telegram/ entry redirects to the site root', async ({ page }) => {
|
test('outside Telegram, the /telegram/ entry redirects to the site root', async ({ page }) => {
|
||||||
await page.goto('/telegram/');
|
await page.goto('/telegram/');
|
||||||
|
|
||||||
|
|||||||
+6
-1
@@ -18,6 +18,7 @@
|
|||||||
import CommsHub from './game/CommsHub.svelte';
|
import CommsHub from './game/CommsHub.svelte';
|
||||||
import Feedback from './screens/Feedback.svelte';
|
import Feedback from './screens/Feedback.svelte';
|
||||||
import Blocked from './screens/Blocked.svelte';
|
import Blocked from './screens/Blocked.svelte';
|
||||||
|
import BootError from './screens/BootError.svelte';
|
||||||
|
|
||||||
onMount(() => {
|
onMount(() => {
|
||||||
void bootstrap();
|
void bootstrap();
|
||||||
@@ -83,6 +84,10 @@
|
|||||||
{#if !routeIsLobby}
|
{#if !routeIsLobby}
|
||||||
<div class="splash">{t('common.loading')}</div>
|
<div class="splash">{t('common.loading')}</div>
|
||||||
{/if}
|
{/if}
|
||||||
|
{:else if app.bootError}
|
||||||
|
<!-- A Mini App launch that failed to authenticate (e.g. the backend was down mid-deploy):
|
||||||
|
show the retry screen instead of falling back to the web login. -->
|
||||||
|
<BootError />
|
||||||
{:else if app.blocked}
|
{:else if app.blocked}
|
||||||
<Blocked />
|
<Blocked />
|
||||||
{:else}
|
{:else}
|
||||||
@@ -123,7 +128,7 @@
|
|||||||
<StaleInviteModal />
|
<StaleInviteModal />
|
||||||
<WelcomeRedeemModal />
|
<WelcomeRedeemModal />
|
||||||
|
|
||||||
{#if routeIsLobby && !app.splashDone && !app.blocked}
|
{#if routeIsLobby && !app.splashDone && !app.blocked && !app.bootError}
|
||||||
<Splash />
|
<Splash />
|
||||||
{/if}
|
{/if}
|
||||||
|
|
||||||
|
|||||||
@@ -18,6 +18,7 @@ import {
|
|||||||
telegramDisableVerticalSwipes,
|
telegramDisableVerticalSwipes,
|
||||||
telegramHaptic,
|
telegramHaptic,
|
||||||
telegramLaunch,
|
telegramLaunch,
|
||||||
|
type TelegramLaunch,
|
||||||
telegramOnEvent,
|
telegramOnEvent,
|
||||||
telegramRequestFullscreen,
|
telegramRequestFullscreen,
|
||||||
telegramSetChrome,
|
telegramSetChrome,
|
||||||
@@ -41,6 +42,10 @@ export interface Toast {
|
|||||||
|
|
||||||
export const app = $state<{
|
export const app = $state<{
|
||||||
ready: boolean;
|
ready: boolean;
|
||||||
|
/** Inside a Mini App, set when the launch failed to authenticate after its retries (e.g. the
|
||||||
|
* backend was down during a deploy). App.svelte then renders the boot-error retry screen
|
||||||
|
* instead of the web login — a Mini App has no manual sign-in to fall back to. */
|
||||||
|
bootError: boolean;
|
||||||
/** Whether the lobby's first cold load has settled (success or error). The loading splash
|
/** Whether the lobby's first cold load has settled (success or error). The loading splash
|
||||||
* (components/Splash.svelte) watches it to know when to dismiss; set by screens/Lobby. */
|
* (components/Splash.svelte) watches it to know when to dismiss; set by screens/Lobby. */
|
||||||
lobbyReady: boolean;
|
lobbyReady: boolean;
|
||||||
@@ -90,6 +95,7 @@ export const app = $state<{
|
|||||||
resync: number;
|
resync: number;
|
||||||
}>({
|
}>({
|
||||||
ready: false,
|
ready: false,
|
||||||
|
bootError: false,
|
||||||
lobbyReady: false,
|
lobbyReady: false,
|
||||||
splashDone: false,
|
splashDone: false,
|
||||||
streamAlive: false,
|
streamAlive: false,
|
||||||
@@ -563,14 +569,7 @@ export async function bootstrap(): Promise<void> {
|
|||||||
// listener above then re-syncs the safe-area insets. Desktop keeps the bot's full-size
|
// listener above then re-syncs the safe-area insets. Desktop keeps the bot's full-size
|
||||||
// window. No-op on clients predating Bot API 8.0.
|
// window. No-op on clients predating Bot API 8.0.
|
||||||
telegramRequestFullscreen();
|
telegramRequestFullscreen();
|
||||||
try {
|
await bootTelegram(launch);
|
||||||
await adoptSession(await gateway.authTelegram(launch.initData));
|
|
||||||
// A blocked account skips deep-link routing — the blocked screen overlays every route.
|
|
||||||
if (!app.blocked) await routeStartParam(launch.startParam);
|
|
||||||
} catch (err) {
|
|
||||||
handleError(err);
|
|
||||||
navigate('/login');
|
|
||||||
}
|
|
||||||
app.ready = true;
|
app.ready = true;
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -585,6 +584,57 @@ export async function bootstrap(): Promise<void> {
|
|||||||
app.ready = true;
|
app.ready = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Inside a Mini App the only identity is the Telegram session, so a failed launch must never fall
|
||||||
|
// back to the web login screen. A transient backend outage (a deploy rolling over) is retried a
|
||||||
|
// few times in silence; only then does the boot-error screen surface, from which Retry re-runs the
|
||||||
|
// same path (retryTelegramBoot).
|
||||||
|
const TELEGRAM_BOOT_RETRIES = 2;
|
||||||
|
const TELEGRAM_BOOT_RETRY_MS = 1200;
|
||||||
|
|
||||||
|
function delay(ms: number): Promise<void> {
|
||||||
|
return new Promise((resolve) => setTimeout(resolve, ms));
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* bootTelegram authenticates a Mini App launch from its initData and routes any deep-link start
|
||||||
|
* parameter, retrying a few times on a transient failure before raising the boot-error screen
|
||||||
|
* (app.bootError). A blocked account is terminal — it switches straight to the blocked screen
|
||||||
|
* without retrying.
|
||||||
|
*/
|
||||||
|
async function bootTelegram(launch: TelegramLaunch): Promise<void> {
|
||||||
|
for (let attempt = 0; ; attempt++) {
|
||||||
|
try {
|
||||||
|
await adoptSession(await gateway.authTelegram(launch.initData));
|
||||||
|
// A blocked account skips deep-link routing — the blocked screen overlays every route.
|
||||||
|
if (!app.blocked) await routeStartParam(launch.startParam);
|
||||||
|
app.bootError = false;
|
||||||
|
return;
|
||||||
|
} catch (err) {
|
||||||
|
if (err instanceof GatewayError && err.code === 'account_blocked') {
|
||||||
|
await enterBlocked();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
if (attempt >= TELEGRAM_BOOT_RETRIES) {
|
||||||
|
app.bootError = true;
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
await delay(TELEGRAM_BOOT_RETRY_MS);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* retryTelegramBoot re-attempts the Mini App launch from the boot-error screen's Retry button. It
|
||||||
|
* clears the error and shows the loading state again, then runs the same retrying boot; on success
|
||||||
|
* the app renders normally, otherwise the boot-error screen returns.
|
||||||
|
*/
|
||||||
|
export async function retryTelegramBoot(): Promise<void> {
|
||||||
|
app.bootError = false;
|
||||||
|
app.ready = false;
|
||||||
|
await bootTelegram(telegramLaunch());
|
||||||
|
app.ready = true;
|
||||||
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* routeStartParam navigates a Telegram deep-link start parameter to its target: a
|
* routeStartParam navigates a Telegram deep-link start parameter to its target: a
|
||||||
* specific game, the friends screen with a friend-code redemption, or the lobby
|
* specific game, the friends screen with a friend-code redemption, or the lobby
|
||||||
|
|||||||
@@ -11,6 +11,9 @@ export const en = {
|
|||||||
'blocked.temporary': 'Your account is blocked until {until}.',
|
'blocked.temporary': 'Your account is blocked until {until}.',
|
||||||
'blocked.reason': 'Reason:',
|
'blocked.reason': 'Reason:',
|
||||||
|
|
||||||
|
'boot.errorTitle': "Couldn't load the game",
|
||||||
|
'boot.errorBody': 'Please try again in a moment.',
|
||||||
|
|
||||||
'common.back': 'Back',
|
'common.back': 'Back',
|
||||||
'common.cancel': 'Cancel',
|
'common.cancel': 'Cancel',
|
||||||
'common.ok': 'OK',
|
'common.ok': 'OK',
|
||||||
|
|||||||
@@ -12,6 +12,9 @@ export const ru: Record<MessageKey, string> = {
|
|||||||
'blocked.temporary': 'Ваша учётная запись заблокирована до {until}.',
|
'blocked.temporary': 'Ваша учётная запись заблокирована до {until}.',
|
||||||
'blocked.reason': 'Причина:',
|
'blocked.reason': 'Причина:',
|
||||||
|
|
||||||
|
'boot.errorTitle': 'Не удалось загрузить игру',
|
||||||
|
'boot.errorBody': 'Попробуйте ещё раз или зайдите позже.',
|
||||||
|
|
||||||
'common.back': 'Назад',
|
'common.back': 'Назад',
|
||||||
'common.cancel': 'Отмена',
|
'common.cancel': 'Отмена',
|
||||||
'common.ok': 'ОК',
|
'common.ok': 'ОК',
|
||||||
|
|||||||
@@ -136,7 +136,10 @@ export class MockGateway implements GatewayClient {
|
|||||||
}
|
}
|
||||||
|
|
||||||
// --- auth ---
|
// --- auth ---
|
||||||
async authTelegram(): Promise<Session> {
|
async authTelegram(initData: string): Promise<Session> {
|
||||||
|
// e2e hook: an initData carrying this sentinel simulates a backend that rejects the launch,
|
||||||
|
// so the Mini App boot-failure path (silent retries → boot-error screen) can be exercised.
|
||||||
|
if (initData.includes('bootfail')) throw new GatewayError('unavailable');
|
||||||
return { ...SESSION, isGuest: false };
|
return { ...SESSION, isGuest: false };
|
||||||
}
|
}
|
||||||
async authGuest(): Promise<Session> {
|
async authGuest(): Promise<Session> {
|
||||||
|
|||||||
@@ -0,0 +1,63 @@
|
|||||||
|
<script lang="ts">
|
||||||
|
// The Mini App launch failed to authenticate after its silent retries (e.g. the backend was
|
||||||
|
// briefly down during a deploy). Inside Telegram there is no web login to fall back to, so this
|
||||||
|
// terminal screen offers a manual Retry that re-runs the launch (app.svelte retryTelegramBoot).
|
||||||
|
import { retryTelegramBoot } from '../lib/app.svelte';
|
||||||
|
import { t } from '../lib/i18n/index.svelte';
|
||||||
|
|
||||||
|
let retrying = $state(false);
|
||||||
|
async function retry(): Promise<void> {
|
||||||
|
if (retrying) return;
|
||||||
|
retrying = true;
|
||||||
|
try {
|
||||||
|
await retryTelegramBoot();
|
||||||
|
} finally {
|
||||||
|
retrying = false;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
</script>
|
||||||
|
|
||||||
|
<div class="boot">
|
||||||
|
<div class="card">
|
||||||
|
<h1>{t('boot.errorTitle')}</h1>
|
||||||
|
<p class="msg">{t('boot.errorBody')}</p>
|
||||||
|
<button class="retry" onclick={retry} disabled={retrying}>{t('common.retry')}</button>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
|
||||||
|
<style>
|
||||||
|
.boot {
|
||||||
|
height: 100%;
|
||||||
|
display: grid;
|
||||||
|
place-items: center;
|
||||||
|
padding: 24px;
|
||||||
|
background: var(--bg);
|
||||||
|
}
|
||||||
|
.card {
|
||||||
|
max-width: 28rem;
|
||||||
|
display: flex;
|
||||||
|
flex-direction: column;
|
||||||
|
align-items: center;
|
||||||
|
gap: 1rem;
|
||||||
|
text-align: center;
|
||||||
|
color: var(--text);
|
||||||
|
}
|
||||||
|
h1 {
|
||||||
|
margin: 0;
|
||||||
|
font-size: 1.25rem;
|
||||||
|
}
|
||||||
|
.msg {
|
||||||
|
margin: 0;
|
||||||
|
color: var(--text-muted);
|
||||||
|
}
|
||||||
|
.retry {
|
||||||
|
padding: 9px 16px;
|
||||||
|
border: 1px solid var(--accent);
|
||||||
|
background: var(--accent);
|
||||||
|
color: var(--accent-text);
|
||||||
|
border-radius: var(--radius-sm);
|
||||||
|
}
|
||||||
|
.retry:disabled {
|
||||||
|
opacity: 0.5;
|
||||||
|
}
|
||||||
|
</style>
|
||||||
@@ -244,9 +244,10 @@
|
|||||||
</form>
|
</form>
|
||||||
{/if}
|
{/if}
|
||||||
|
|
||||||
<!-- Linking & merge. Shown to everyone, including guests, who
|
<!-- Linking & merge. Hidden for now: we target provider sign-in, and the anonymous
|
||||||
upgrade by binding their first identity. -->
|
/app/ guest (whose upgrade path this is) comes later. Kept wired — drop `hidden`
|
||||||
<section class="emailbox">
|
to re-enable, together with the skipped linking specs in e2e/social.spec.ts. -->
|
||||||
|
<section class="emailbox" hidden>
|
||||||
<h3>{t('profile.linkAccount')}</h3>
|
<h3>{t('profile.linkAccount')}</h3>
|
||||||
{#if !emailSent}
|
{#if !emailSent}
|
||||||
<div class="addrow">
|
<div class="addrow">
|
||||||
|
|||||||
Reference in New Issue
Block a user