fix(deploy): migrate after stop and shrink DB pool
Deploy / release (push) Skipped
Deploy / deploy (push) Successful in 1m52s

Stage migrate hit ER_CON_COUNT_ERROR while live still held the pool. Build in stage with DATABASE_POOL_SIZE=5, run db:migrate only after stopping the service (with retries), and lower the default pool from 40 to 10.

Co-authored-by: Cursor <[email protected]>
This commit is contained in:
SimoandCursor committed 2026-07-21 21:44:06 +02:00
1 parent 687e1f9fb0
commit fa40eebeed
4 files changed
+34 -7

No files matched your search

+1 -1
View File
@@ -4,7 +4,7 @@
DATABASE_URL=mysql://user:[email protected]:3306/atomcms
# Optional pool tuning (defaults shown)
DATABASE_POOL_SIZE=40
DATABASE_POOL_SIZE=10
DATABASE_IDLE_TIMEOUT_MS=300000
DATABASE_CONNECT_TIMEOUT_MS=10000
+26 -3
View File
@@ -293,9 +293,13 @@ jobs:
find . -maxdepth 3 -name '*.tsbuildinfo' -delete 2>/dev/null || true
rm -rf .output dist .next .next/types .next/dev
# Stage shares MySQL with the live app + emulator. Keep the stage pool
# tiny so install/test/build cannot exhaust max_connections.
export DATABASE_POOL_SIZE="${DEPLOY_DATABASE_POOL_SIZE:-5}"
echo "STAGE DATABASE_POOL_SIZE=${DATABASE_POOL_SIZE}"
pnpm install --frozen-lockfile
# Additive migrations while the old build still serves traffic.
pnpm db:migrate
# prisma generate does not need a live DB connection.
pnpm prisma:generate
pnpm typecheck
pnpm test
@@ -308,10 +312,29 @@ jobs:
exit 1
fi
echo "Cutover: stop service, sync code, swap .next artifact..."
echo "Cutover: stop service (free DB connections), migrate, swap .next..."
CUTOVER_STARTED=1
sudo systemctl stop atom-nexst.service || true
pkill -f 'next-server' || true
# Give MariaDB a moment to reclaim the live pool.
sleep 2
# Migrate only after live is stopped — avoids ER_CON_COUNT_ERROR while
# the old process still holds DATABASE_POOL_SIZE connections.
cd "${STAGE}"
MIGRATE_OK=0
for i in 1 2 3 4 5; do
if pnpm db:migrate; then
MIGRATE_OK=1
break
fi
echo "migrate attempt ${i}/5 failed (likely DB connections), retrying..."
sleep 3
done
if [ "${MIGRATE_OK}" != "1" ]; then
echo "ERROR: db:migrate failed after retries" >&2
exit 1
fi
cd "${LIVE}"
echo "Hard reset live tree to origin/main (no nuclear src wipe)..."
+1 -1
View File
@@ -8,7 +8,7 @@ const schema = z.object({
.enum(["development", "test", "production"])
.default("development"),
DATABASE_URL: z.string().url(),
DATABASE_POOL_SIZE: z.coerce.number().int().positive().default(40),
DATABASE_POOL_SIZE: z.coerce.number().int().positive().default(10),
DATABASE_IDLE_TIMEOUT_MS: z.coerce.number().int().positive().default(300_000),
DATABASE_CONNECT_TIMEOUT_MS: z.coerce
.number()
+6 -2
View File
@@ -14,6 +14,7 @@ describe("production deploy workflow", () => {
expect(deployJob).toContain("/var/tmp/atom-nexst-stage-");
expect(deployJob).toContain('ln -sfn "${LIVE}/.env"');
expect(deployJob).toContain("-e storage");
expect(deployJob).toContain("DATABASE_POOL_SIZE=");
expect(deployJob).not.toContain("SKIP_ENV_VALIDATION=1");
expect(deployJob).toContain("pnpm install --frozen-lockfile");
});
@@ -25,7 +26,7 @@ describe("production deploy workflow", () => {
const reclaimAt = deployJob.indexOf(
'sudo chown -R "$' + "{DEPLOY_USER}:" + '$' + '{DEPLOY_GROUP}"',
);
const fetchAt = deployJob.indexOf("git -C \"${LIVE}\" fetch origin --prune");
const fetchAt = deployJob.indexOf('git -C "${LIVE}" fetch origin --prune');
expect(reclaimAt).toBeGreaterThan(-1);
expect(fetchAt).toBeGreaterThan(reclaimAt);
});
@@ -48,6 +49,7 @@ describe("production deploy workflow", () => {
expect(deployJob).toContain("Rolling back .next to previous artifact");
const buildAt = deployJob.indexOf("pnpm build");
const stopAt = deployJob.indexOf("sudo systemctl stop atom-nexst.service");
const migrateAt = deployJob.indexOf("pnpm db:migrate");
const startLabelAt = deployJob.indexOf("Starting systemd service...");
const startAt = deployJob.indexOf(
"sudo systemctl start atom-nexst.service",
@@ -55,7 +57,9 @@ describe("production deploy workflow", () => {
);
expect(buildAt).toBeGreaterThan(-1);
expect(stopAt).toBeGreaterThan(buildAt);
expect(startLabelAt).toBeGreaterThan(stopAt);
// Migrate runs after stop so the live pool frees DB slots.
expect(migrateAt).toBeGreaterThan(stopAt);
expect(startLabelAt).toBeGreaterThan(migrateAt);
expect(startAt).toBeGreaterThan(startLabelAt);
});