feat(ops): self-heal disk pressure instead of only alerting
The 5-minute disk probe now reclaims storage automatically: from 85% it runs the gentle age-windowed Docker prune, from 90% it drops the age windows (docker-prune.sh --force: all unused build cache and unreferenced images, all stopped containers) so a mount can never silently max out. Alerts still fire at 85/90/95% and their hint now points at non-Docker growth when reclaiming is not enough. Force mode is reserved for the worker; deploys keep the gentle mode. Volumes are off-limits in every path.
This commit is contained in:
1 parent
fe26ca3ff3
commit
1caef76f82
4 files changed
+73
-18
No files matched your search
+30
-13
@@ -1,15 +1,21 @@
|
||||
#!/usr/bin/env bash
|
||||
# Reclaim Docker's unused cache so host storage stays bounded.
|
||||
#
|
||||
# Safe scopes only, by design:
|
||||
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store is
|
||||
# shared across builds; everything newer than that speeds up rebuilds).
|
||||
# - Images referenced by NO running/stopped container and older than 7 days
|
||||
# (covers stale epicnext-cms sha tags, old mariadb/byparr pulls, etc.).
|
||||
# - Containers stopped for more than 24h.
|
||||
# Modes:
|
||||
# (default) — gentle, age-windowed (keeps rollback + rebuild speed):
|
||||
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store
|
||||
# is shared across builds; everything newer speeds up rebuilds).
|
||||
# - Images referenced by NO container and older than 7 days.
|
||||
# - Containers stopped for more than 24h.
|
||||
# --force — emergency mode ("never let the disk max out"): drops every age
|
||||
# window and reclaims all unused bytes Docker can free:
|
||||
# - ALL unreferenced build cache,
|
||||
# - ALL unreferenced images (no 7-day grace),
|
||||
# - ALL stopped containers.
|
||||
# Trade-off: waiting rebuilds re-fetch deps/images later.
|
||||
#
|
||||
# Volumes are NEVER pruned here: mariadb-turbo-data is a database. This script
|
||||
# is idempotent and exits 0 when Docker is unavailable.
|
||||
# Volumes are NEVER pruned in either mode: mariadb-turbo-data is a database.
|
||||
# Idempotent; exits 0 when Docker is unavailable.
|
||||
set -Eeuo pipefail
|
||||
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
LOG_DIR="${LOG_DIR:-$DIR/logs}"
|
||||
@@ -17,17 +23,28 @@ mkdir -p "$LOG_DIR"
|
||||
LOG_FILE="$LOG_DIR/docker-prune.log"
|
||||
now() { date '+%Y-%m-%d %H:%M:%S'; }
|
||||
|
||||
FORCE=0
|
||||
if [[ "${1:-}" == "--force" ]]; then
|
||||
FORCE=1
|
||||
fi
|
||||
|
||||
command -v docker >/dev/null 2>&1 || {
|
||||
printf '[%s] docker CLI unavailable; nothing to prune\n' "$(now)" >>"$LOG_FILE"
|
||||
exit 0
|
||||
}
|
||||
|
||||
printf '\n[%s] === docker prune start ===\n' "$(now)" >>"$LOG_FILE"
|
||||
printf '\n[%s] === docker prune start%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
|
||||
docker system df >>"$LOG_FILE" 2>&1 || true
|
||||
|
||||
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
|
||||
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
|
||||
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
|
||||
if (( FORCE )); then
|
||||
docker builder prune -af >>"$LOG_FILE" 2>&1 || true
|
||||
docker image prune -af >>"$LOG_FILE" 2>&1 || true
|
||||
docker container prune -f >>"$LOG_FILE" 2>&1 || true
|
||||
else
|
||||
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
|
||||
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
|
||||
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
|
||||
fi
|
||||
|
||||
printf '\n[%s] === docker prune complete ===\n' "$(now)" >>"$LOG_FILE"
|
||||
printf '\n[%s] === docker prune complete%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
|
||||
docker system df >>"$LOG_FILE" 2>&1 || true
|
||||
+36
-4
@@ -17,7 +17,11 @@ import {
|
||||
healthDegraded,
|
||||
} from "../src/lib/services/alert";
|
||||
import { runCatalogExport } from "../src/lib/services/catalog-git-export";
|
||||
import { diskLevel, parseDfOutput } from "../src/lib/services/disk-usage";
|
||||
import {
|
||||
DISK_THRESHOLDS,
|
||||
diskLevel,
|
||||
parseDfOutput,
|
||||
} from "../src/lib/services/disk-usage";
|
||||
import { invalidateNewsCache } from "../src/lib/services/news-cache";
|
||||
import { rcon } from "../src/lib/services/rcon";
|
||||
|
||||
@@ -133,6 +137,30 @@ async function checkDiskUsage(): Promise<void> {
|
||||
});
|
||||
await diskPressure(usage);
|
||||
}
|
||||
|
||||
// Self-healing: never let a mount max out. Safe prune from the warning
|
||||
// mark, forced prune (drop age windows) from the error mark. Cooldown is
|
||||
// keyed separately so a stuck fill level does not re-prune every run
|
||||
// while still escalating to the forced path on real pressure.
|
||||
if (usage.percent >= DISK_THRESHOLDS.error) {
|
||||
if (canAlert(`disk-prune-force-${usage.mount}`)) {
|
||||
logger.warn("Disk near full — forcing Docker cache reclaim", {
|
||||
module: "jobs",
|
||||
mount: usage.mount,
|
||||
percent: usage.percent,
|
||||
});
|
||||
await pruneDockerCache(true);
|
||||
}
|
||||
} else if (usage.percent >= DISK_THRESHOLDS.warning) {
|
||||
if (canAlert(`disk-prune-${usage.mount}`)) {
|
||||
logger.info("Disk at high water mark — running safe Docker prune", {
|
||||
module: "jobs",
|
||||
mount: usage.mount,
|
||||
percent: usage.percent,
|
||||
});
|
||||
await pruneDockerCache(false);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -282,8 +310,10 @@ async function cleanupOldSessions(): Promise<void> {
|
||||
|
||||
/** Host-side: reclaim Docker's unused cache (build cache, unreferenced images,
|
||||
* stopped containers). Volumes and in-use images are never touched. No-op when
|
||||
* docker or the prune script is unavailable. */
|
||||
async function pruneDockerCache(): Promise<void> {
|
||||
* docker or the prune script is unavailable. With force=true the age windows
|
||||
* are dropped (docker-prune.sh --force) so every unused byte is reclaimed —
|
||||
* the emergency path for a nearly-full disk. */
|
||||
async function pruneDockerCache(force = false): Promise<void> {
|
||||
const { access } = await import("node:fs/promises");
|
||||
const { resolve } = await import("node:path");
|
||||
const { spawn } = await import("node:child_process");
|
||||
@@ -294,7 +324,9 @@ async function pruneDockerCache(): Promise<void> {
|
||||
return;
|
||||
}
|
||||
await new Promise<void>((resolvePromise) => {
|
||||
const child = spawn("bash", [script], { stdio: "ignore" });
|
||||
const child = spawn("bash", [script, ...(force ? ["--force"] : [])], {
|
||||
stdio: "ignore",
|
||||
});
|
||||
child.on("error", (err) =>
|
||||
captureWorkerError(
|
||||
err,
|
||||
|
||||
@@ -36,6 +36,12 @@ it("preserves production runtime configuration and recent cache", () => {
|
||||
);
|
||||
expect(prune).toContain('docker image prune -af --filter "until=168h"');
|
||||
expect(prune).toContain('docker container prune -f --filter "until=24h"');
|
||||
// Emergency `--force` mode drops every age window to reclaim unused bytes,
|
||||
// but even then volumes are off-limits.
|
||||
expect(prune).toContain('== "--force" ]]');
|
||||
expect(prune).toContain("FORCE=1");
|
||||
expect(prune).toContain("(( FORCE ))");
|
||||
expect(deploy).not.toContain("--force");
|
||||
expect(deploy).not.toContain("docker volume prune");
|
||||
expect(prune).not.toContain("docker volume prune");
|
||||
});
|
||||
|
||||
@@ -306,7 +306,7 @@ export function diskPressure(usage: {
|
||||
used: formatBytes(usage.usedBytes),
|
||||
total: formatBytes(usage.totalBytes),
|
||||
available: formatBytes(usage.availableBytes),
|
||||
hint: "Docker prune runs nightly; run scripts/docker-prune.sh to reclaim cache sooner.",
|
||||
hint: "Docker cache reclaim auto-fired with this alert; if the disk is still filling, the growth is outside Docker (check Gamedata/uploads/storage).",
|
||||
},
|
||||
});
|
||||
}
|
||||
Reference in new issue
Block a user