fix(deploy): migrate after stop and shrink DB pool
Stage migrate hit ER_CON_COUNT_ERROR while live still held the pool. Build in stage with DATABASE_POOL_SIZE=5, run db:migrate only after stopping the service (with retries), and lower the default pool from 40 to 10. Co-authored-by: Cursor <[email protected]>
This commit is contained in:
1 parent
687e1f9fb0
commit
fa40eebeed
4 files changed
+34
-7
No files matched your search
+1
-1
@@ -4,7 +4,7 @@
|
|||||||
DATABASE_URL=mysql://user:[email protected]:3306/atomcms
|
DATABASE_URL=mysql://user:[email protected]:3306/atomcms
|
||||||
|
|
||||||
# Optional pool tuning (defaults shown)
|
# Optional pool tuning (defaults shown)
|
||||||
DATABASE_POOL_SIZE=40
|
DATABASE_POOL_SIZE=10
|
||||||
DATABASE_IDLE_TIMEOUT_MS=300000
|
DATABASE_IDLE_TIMEOUT_MS=300000
|
||||||
DATABASE_CONNECT_TIMEOUT_MS=10000
|
DATABASE_CONNECT_TIMEOUT_MS=10000
|
||||||
|
|
||||||
|
|||||||
@@ -293,9 +293,13 @@ jobs:
|
|||||||
find . -maxdepth 3 -name '*.tsbuildinfo' -delete 2>/dev/null || true
|
find . -maxdepth 3 -name '*.tsbuildinfo' -delete 2>/dev/null || true
|
||||||
rm -rf .output dist .next .next/types .next/dev
|
rm -rf .output dist .next .next/types .next/dev
|
||||||
|
|
||||||
|
# Stage shares MySQL with the live app + emulator. Keep the stage pool
|
||||||
|
# tiny so install/test/build cannot exhaust max_connections.
|
||||||
|
export DATABASE_POOL_SIZE="${DEPLOY_DATABASE_POOL_SIZE:-5}"
|
||||||
|
echo "STAGE DATABASE_POOL_SIZE=${DATABASE_POOL_SIZE}"
|
||||||
|
|
||||||
pnpm install --frozen-lockfile
|
pnpm install --frozen-lockfile
|
||||||
# Additive migrations while the old build still serves traffic.
|
# prisma generate does not need a live DB connection.
|
||||||
pnpm db:migrate
|
|
||||||
pnpm prisma:generate
|
pnpm prisma:generate
|
||||||
pnpm typecheck
|
pnpm typecheck
|
||||||
pnpm test
|
pnpm test
|
||||||
@@ -308,10 +312,29 @@ jobs:
|
|||||||
exit 1
|
exit 1
|
||||||
fi
|
fi
|
||||||
|
|
||||||
echo "Cutover: stop service, sync code, swap .next artifact..."
|
echo "Cutover: stop service (free DB connections), migrate, swap .next..."
|
||||||
CUTOVER_STARTED=1
|
CUTOVER_STARTED=1
|
||||||
sudo systemctl stop atom-nexst.service || true
|
sudo systemctl stop atom-nexst.service || true
|
||||||
pkill -f 'next-server' || true
|
pkill -f 'next-server' || true
|
||||||
|
# Give MariaDB a moment to reclaim the live pool.
|
||||||
|
sleep 2
|
||||||
|
|
||||||
|
# Migrate only after live is stopped — avoids ER_CON_COUNT_ERROR while
|
||||||
|
# the old process still holds DATABASE_POOL_SIZE connections.
|
||||||
|
cd "${STAGE}"
|
||||||
|
MIGRATE_OK=0
|
||||||
|
for i in 1 2 3 4 5; do
|
||||||
|
if pnpm db:migrate; then
|
||||||
|
MIGRATE_OK=1
|
||||||
|
break
|
||||||
|
fi
|
||||||
|
echo "migrate attempt ${i}/5 failed (likely DB connections), retrying..."
|
||||||
|
sleep 3
|
||||||
|
done
|
||||||
|
if [ "${MIGRATE_OK}" != "1" ]; then
|
||||||
|
echo "ERROR: db:migrate failed after retries" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
cd "${LIVE}"
|
cd "${LIVE}"
|
||||||
echo "Hard reset live tree to origin/main (no nuclear src wipe)..."
|
echo "Hard reset live tree to origin/main (no nuclear src wipe)..."
|
||||||
|
|||||||
+1
-1
@@ -8,7 +8,7 @@ const schema = z.object({
|
|||||||
.enum(["development", "test", "production"])
|
.enum(["development", "test", "production"])
|
||||||
.default("development"),
|
.default("development"),
|
||||||
DATABASE_URL: z.string().url(),
|
DATABASE_URL: z.string().url(),
|
||||||
DATABASE_POOL_SIZE: z.coerce.number().int().positive().default(40),
|
DATABASE_POOL_SIZE: z.coerce.number().int().positive().default(10),
|
||||||
DATABASE_IDLE_TIMEOUT_MS: z.coerce.number().int().positive().default(300_000),
|
DATABASE_IDLE_TIMEOUT_MS: z.coerce.number().int().positive().default(300_000),
|
||||||
DATABASE_CONNECT_TIMEOUT_MS: z.coerce
|
DATABASE_CONNECT_TIMEOUT_MS: z.coerce
|
||||||
.number()
|
.number()
|
||||||
|
|||||||
@@ -14,6 +14,7 @@ describe("production deploy workflow", () => {
|
|||||||
expect(deployJob).toContain("/var/tmp/atom-nexst-stage-");
|
expect(deployJob).toContain("/var/tmp/atom-nexst-stage-");
|
||||||
expect(deployJob).toContain('ln -sfn "${LIVE}/.env"');
|
expect(deployJob).toContain('ln -sfn "${LIVE}/.env"');
|
||||||
expect(deployJob).toContain("-e storage");
|
expect(deployJob).toContain("-e storage");
|
||||||
|
expect(deployJob).toContain("DATABASE_POOL_SIZE=");
|
||||||
expect(deployJob).not.toContain("SKIP_ENV_VALIDATION=1");
|
expect(deployJob).not.toContain("SKIP_ENV_VALIDATION=1");
|
||||||
expect(deployJob).toContain("pnpm install --frozen-lockfile");
|
expect(deployJob).toContain("pnpm install --frozen-lockfile");
|
||||||
});
|
});
|
||||||
@@ -25,7 +26,7 @@ describe("production deploy workflow", () => {
|
|||||||
const reclaimAt = deployJob.indexOf(
|
const reclaimAt = deployJob.indexOf(
|
||||||
'sudo chown -R "$' + "{DEPLOY_USER}:" + '$' + '{DEPLOY_GROUP}"',
|
'sudo chown -R "$' + "{DEPLOY_USER}:" + '$' + '{DEPLOY_GROUP}"',
|
||||||
);
|
);
|
||||||
const fetchAt = deployJob.indexOf("git -C \"${LIVE}\" fetch origin --prune");
|
const fetchAt = deployJob.indexOf('git -C "${LIVE}" fetch origin --prune');
|
||||||
expect(reclaimAt).toBeGreaterThan(-1);
|
expect(reclaimAt).toBeGreaterThan(-1);
|
||||||
expect(fetchAt).toBeGreaterThan(reclaimAt);
|
expect(fetchAt).toBeGreaterThan(reclaimAt);
|
||||||
});
|
});
|
||||||
@@ -48,6 +49,7 @@ describe("production deploy workflow", () => {
|
|||||||
expect(deployJob).toContain("Rolling back .next to previous artifact");
|
expect(deployJob).toContain("Rolling back .next to previous artifact");
|
||||||
const buildAt = deployJob.indexOf("pnpm build");
|
const buildAt = deployJob.indexOf("pnpm build");
|
||||||
const stopAt = deployJob.indexOf("sudo systemctl stop atom-nexst.service");
|
const stopAt = deployJob.indexOf("sudo systemctl stop atom-nexst.service");
|
||||||
|
const migrateAt = deployJob.indexOf("pnpm db:migrate");
|
||||||
const startLabelAt = deployJob.indexOf("Starting systemd service...");
|
const startLabelAt = deployJob.indexOf("Starting systemd service...");
|
||||||
const startAt = deployJob.indexOf(
|
const startAt = deployJob.indexOf(
|
||||||
"sudo systemctl start atom-nexst.service",
|
"sudo systemctl start atom-nexst.service",
|
||||||
@@ -55,7 +57,9 @@ describe("production deploy workflow", () => {
|
|||||||
);
|
);
|
||||||
expect(buildAt).toBeGreaterThan(-1);
|
expect(buildAt).toBeGreaterThan(-1);
|
||||||
expect(stopAt).toBeGreaterThan(buildAt);
|
expect(stopAt).toBeGreaterThan(buildAt);
|
||||||
expect(startLabelAt).toBeGreaterThan(stopAt);
|
// Migrate runs after stop so the live pool frees DB slots.
|
||||||
|
expect(migrateAt).toBeGreaterThan(stopAt);
|
||||||
|
expect(startLabelAt).toBeGreaterThan(migrateAt);
|
||||||
expect(startAt).toBeGreaterThan(startLabelAt);
|
expect(startAt).toBeGreaterThan(startLabelAt);
|
||||||
});
|
});
|
||||||
|
|
||||||
|
|||||||
Reference in new issue
Block a user