feat(ops): self-heal disk pressure instead of only alerting
CI / check (push) Successful in 2m36s
CI / deploy (push) Successful in 1m28s
CI / publish-container (push) Successful in 46s

The 5-minute disk probe now reclaims storage automatically: from 85% it runs the gentle age-windowed Docker prune, from 90% it drops the age windows (docker-prune.sh --force: all unused build cache and unreferenced images, all stopped containers) so a mount can never silently max out. Alerts still fire at 85/90/95% and their hint now points at non-Docker growth when reclaiming is not enough. Force mode is reserved for the worker; deploys keep the gentle mode. Volumes are off-limits in every path.
This commit is contained in:
openhands committed 2026-09-13 13:18:00 +02:00
1 parent fe26ca3ff3
commit 1caef76f82
4 files changed
+73 -18

No files matched your search

+30 -13
View File
@@ -1,15 +1,21 @@
#!/usr/bin/env bash
# Reclaim Docker's unused cache so host storage stays bounded.
#
# Safe scopes only, by design:
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store is
# shared across builds; everything newer than that speeds up rebuilds).
# - Images referenced by NO running/stopped container and older than 7 days
# (covers stale epicnext-cms sha tags, old mariadb/byparr pulls, etc.).
# - Containers stopped for more than 24h.
# Modes:
# (default) — gentle, age-windowed (keeps rollback + rebuild speed):
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store
# is shared across builds; everything newer speeds up rebuilds).
# - Images referenced by NO container and older than 7 days.
# - Containers stopped for more than 24h.
# --force — emergency mode ("never let the disk max out"): drops every age
# window and reclaims all unused bytes Docker can free:
# - ALL unreferenced build cache,
# - ALL unreferenced images (no 7-day grace),
# - ALL stopped containers.
# Trade-off: waiting rebuilds re-fetch deps/images later.
#
# Volumes are NEVER pruned here: mariadb-turbo-data is a database. This script
# is idempotent and exits 0 when Docker is unavailable.
# Volumes are NEVER pruned in either mode: mariadb-turbo-data is a database.
# Idempotent; exits 0 when Docker is unavailable.
set -Eeuo pipefail
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
LOG_DIR="${LOG_DIR:-$DIR/logs}"
@@ -17,17 +23,28 @@ mkdir -p "$LOG_DIR"
LOG_FILE="$LOG_DIR/docker-prune.log"
now() { date '+%Y-%m-%d %H:%M:%S'; }
FORCE=0
if [[ "${1:-}" == "--force" ]]; then
FORCE=1
fi
command -v docker >/dev/null 2>&1 || {
printf '[%s] docker CLI unavailable; nothing to prune\n' "$(now)" >>"$LOG_FILE"
exit 0
}
printf '\n[%s] === docker prune start ===\n' "$(now)" >>"$LOG_FILE"
printf '\n[%s] === docker prune start%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
docker system df >>"$LOG_FILE" 2>&1 || true
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
if (( FORCE )); then
docker builder prune -af >>"$LOG_FILE" 2>&1 || true
docker image prune -af >>"$LOG_FILE" 2>&1 || true
docker container prune -f >>"$LOG_FILE" 2>&1 || true
else
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
fi
printf '\n[%s] === docker prune complete ===\n' "$(now)" >>"$LOG_FILE"
printf '\n[%s] === docker prune complete%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
docker system df >>"$LOG_FILE" 2>&1 || true
+36 -4
View File
@@ -17,7 +17,11 @@ import {
healthDegraded,
} from "../src/lib/services/alert";
import { runCatalogExport } from "../src/lib/services/catalog-git-export";
import { diskLevel, parseDfOutput } from "../src/lib/services/disk-usage";
import {
DISK_THRESHOLDS,
diskLevel,
parseDfOutput,
} from "../src/lib/services/disk-usage";
import { invalidateNewsCache } from "../src/lib/services/news-cache";
import { rcon } from "../src/lib/services/rcon";
@@ -133,6 +137,30 @@ async function checkDiskUsage(): Promise<void> {
});
await diskPressure(usage);
}
// Self-healing: never let a mount max out. Safe prune from the warning
// mark, forced prune (drop age windows) from the error mark. Cooldown is
// keyed separately so a stuck fill level does not re-prune every run
// while still escalating to the forced path on real pressure.
if (usage.percent >= DISK_THRESHOLDS.error) {
if (canAlert(`disk-prune-force-${usage.mount}`)) {
logger.warn("Disk near full — forcing Docker cache reclaim", {
module: "jobs",
mount: usage.mount,
percent: usage.percent,
});
await pruneDockerCache(true);
}
} else if (usage.percent >= DISK_THRESHOLDS.warning) {
if (canAlert(`disk-prune-${usage.mount}`)) {
logger.info("Disk at high water mark — running safe Docker prune", {
module: "jobs",
mount: usage.mount,
percent: usage.percent,
});
await pruneDockerCache(false);
}
}
}
}
@@ -282,8 +310,10 @@ async function cleanupOldSessions(): Promise<void> {
/** Host-side: reclaim Docker's unused cache (build cache, unreferenced images,
* stopped containers). Volumes and in-use images are never touched. No-op when
* docker or the prune script is unavailable. */
async function pruneDockerCache(): Promise<void> {
* docker or the prune script is unavailable. With force=true the age windows
* are dropped (docker-prune.sh --force) so every unused byte is reclaimed —
* the emergency path for a nearly-full disk. */
async function pruneDockerCache(force = false): Promise<void> {
const { access } = await import("node:fs/promises");
const { resolve } = await import("node:path");
const { spawn } = await import("node:child_process");
@@ -294,7 +324,9 @@ async function pruneDockerCache(): Promise<void> {
return;
}
await new Promise<void>((resolvePromise) => {
const child = spawn("bash", [script], { stdio: "ignore" });
const child = spawn("bash", [script, ...(force ? ["--force"] : [])], {
stdio: "ignore",
});
child.on("error", (err) =>
captureWorkerError(
err,
+6
View File
@@ -36,6 +36,12 @@ it("preserves production runtime configuration and recent cache", () => {
);
expect(prune).toContain('docker image prune -af --filter "until=168h"');
expect(prune).toContain('docker container prune -f --filter "until=24h"');
// Emergency `--force` mode drops every age window to reclaim unused bytes,
// but even then volumes are off-limits.
expect(prune).toContain('== "--force" ]]');
expect(prune).toContain("FORCE=1");
expect(prune).toContain("(( FORCE ))");
expect(deploy).not.toContain("--force");
expect(deploy).not.toContain("docker volume prune");
expect(prune).not.toContain("docker volume prune");
});
+1 -1
View File
@@ -306,7 +306,7 @@ export function diskPressure(usage: {
used: formatBytes(usage.usedBytes),
total: formatBytes(usage.totalBytes),
available: formatBytes(usage.availableBytes),
hint: "Docker prune runs nightly; run scripts/docker-prune.sh to reclaim cache sooner.",
hint: "Docker cache reclaim auto-fired with this alert; if the disk is still filling, the growth is outside Docker (check Gamedata/uploads/storage).",
},
});
}