From 01e22944713612baab4d27fb3be6fb0140301fdc Mon Sep 17 00:00:00 2001 From: Bryan Van Deusen Date: Thu, 8 Oct 2026 20:18:49 -0400 Subject: [PATCH] ci: probe and relax the integration Postgres over TCP, with retries (#5426) On first boot the postgres image runs initdb under a temporary server that listens on the unix socket only, then stops it. The socket readiness probe could catch that server, and the fsync relax then hit 'the database system is shutting down', leaving the suite ~15x slower on internal/api. Only the real server listens on TCP, so probe and relax via -h 127.0.0.1, retry the relax, and print the settings that took effect. Co-Authored-By: Claude Opus 5.5 --- .gitea/workflows/release.yml | 35 +++++++++++++++++++++++++++-------- 1 file changed, 27 insertions(+), 8 deletions(-) diff --git a/.gitea/workflows/release.yml b/.gitea/workflows/release.yml index 492fc566..a8c3a1f4 100644 --- a/.gitea/workflows/release.yml +++ b/.gitea/workflows/release.yml @@ -239,9 +239,16 @@ jobs: # container itself: the run: shell is dash (rule 81), where the # old `/dev/tcp` probe never connects and the loop silently # burned its full two minutes on every run. + # + # Over TCP (-h 127.0.0.1), not the unix socket (#5426). On first + # boot the image's entrypoint runs initdb under a temporary server + # that listens on the socket only, then stops it and starts the + # real one. A socket probe can catch the temporary server, and the + # next step then meets "the database system is shutting down". + # Only the real server listens on TCP. ready="" for i in $(seq 1 60); do - if docker exec "$PG_ID" pg_isready -U minstrel -d minstrel_test -q; then ready=1; break; fi + if docker exec "$PG_ID" pg_isready -h 127.0.0.1 -U minstrel -d minstrel_test -q; then ready=1; break; fi sleep 2 done test -n "$ready" || { echo "FATAL: postgres never became ready"; exit 1; } @@ -259,13 +266,25 @@ jobs: # - fsync / full_page_writes are sighup GUCs and # synchronous_commit is user-context, so pg_reload_conf() picks # all three up with no restart. - # Non-fatal: a perms surprise degrades to "slower", never red CI. - docker exec "$PG_ID" psql -U minstrel -d minstrel_test \ - -c "ALTER SYSTEM SET fsync = off" \ - -c "ALTER SYSTEM SET synchronous_commit = off" \ - -c "ALTER SYSTEM SET full_page_writes = off" \ - -c "SELECT pg_reload_conf()" \ - || echo "WARN: durability relax failed; continuing" + # Non-fatal: a failure degrades to "slower", never red CI. But + # slower is ~15x on internal/api (#5426), so retry a few times and + # print what took effect. + relaxed="" + for i in 1 2 3 4 5; do + if docker exec -e PGPASSWORD=minstrel "$PG_ID" psql -h 127.0.0.1 -U minstrel -d minstrel_test \ + -c "ALTER SYSTEM SET fsync = off" \ + -c "ALTER SYSTEM SET synchronous_commit = off" \ + -c "ALTER SYSTEM SET full_page_writes = off" \ + -c "SELECT pg_reload_conf()"; then relaxed=1; break; fi + sleep 2 + done + if [ -n "$relaxed" ]; then + docker exec -e PGPASSWORD=minstrel "$PG_ID" psql -h 127.0.0.1 -U minstrel -d minstrel_test -At \ + -c "SELECT 'durability: ' || string_agg(name || '=' || setting, ' ' ORDER BY name) FROM pg_settings WHERE name IN ('fsync','synchronous_commit','full_page_writes')" \ + || true + else + echo "::warning::durability relax failed after 5 tries; the suite will run several times slower" + fi # Apply embedded migrations to the fresh test DB, then run the # full suite (no -short → integration tests execute). -p 1: