feat(ops): self-heal disk pressure instead of only alerting
The 5-minute disk probe now reclaims storage automatically: from 85% it runs the gentle age-windowed Docker prune, from 90% it drops the age windows (docker-prune.sh --force: all unused build cache and unreferenced images, all stopped containers) so a mount can never silently max out. Alerts still fire at 85/90/95% and their hint now points at non-Docker growth when reclaiming is not enough. Force mode is reserved for the worker; deploys keep the gentle mode. Volumes are off-limits in every path.
This commit is contained in:
1 parent
fe26ca3ff3
commit
1caef76f82
4 files changed
+73
-18
No files matched your search
+30
-13
@@ -1,15 +1,21 @@
|
|||||||
#!/usr/bin/env bash
|
#!/usr/bin/env bash
|
||||||
# Reclaim Docker's unused cache so host storage stays bounded.
|
# Reclaim Docker's unused cache so host storage stays bounded.
|
||||||
#
|
#
|
||||||
# Safe scopes only, by design:
|
# Modes:
|
||||||
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store is
|
# (default) — gentle, age-windowed (keeps rollback + rebuild speed):
|
||||||
# shared across builds; everything newer than that speeds up rebuilds).
|
# - BuildKit cache older than 72h, hard-capped at 4 GB (Debian /pnpm store
|
||||||
# - Images referenced by NO running/stopped container and older than 7 days
|
# is shared across builds; everything newer speeds up rebuilds).
|
||||||
# (covers stale epicnext-cms sha tags, old mariadb/byparr pulls, etc.).
|
# - Images referenced by NO container and older than 7 days.
|
||||||
# - Containers stopped for more than 24h.
|
# - Containers stopped for more than 24h.
|
||||||
|
# --force — emergency mode ("never let the disk max out"): drops every age
|
||||||
|
# window and reclaims all unused bytes Docker can free:
|
||||||
|
# - ALL unreferenced build cache,
|
||||||
|
# - ALL unreferenced images (no 7-day grace),
|
||||||
|
# - ALL stopped containers.
|
||||||
|
# Trade-off: waiting rebuilds re-fetch deps/images later.
|
||||||
#
|
#
|
||||||
# Volumes are NEVER pruned here: mariadb-turbo-data is a database. This script
|
# Volumes are NEVER pruned in either mode: mariadb-turbo-data is a database.
|
||||||
# is idempotent and exits 0 when Docker is unavailable.
|
# Idempotent; exits 0 when Docker is unavailable.
|
||||||
set -Eeuo pipefail
|
set -Eeuo pipefail
|
||||||
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||||
LOG_DIR="${LOG_DIR:-$DIR/logs}"
|
LOG_DIR="${LOG_DIR:-$DIR/logs}"
|
||||||
@@ -17,17 +23,28 @@ mkdir -p "$LOG_DIR"
|
|||||||
LOG_FILE="$LOG_DIR/docker-prune.log"
|
LOG_FILE="$LOG_DIR/docker-prune.log"
|
||||||
now() { date '+%Y-%m-%d %H:%M:%S'; }
|
now() { date '+%Y-%m-%d %H:%M:%S'; }
|
||||||
|
|
||||||
|
FORCE=0
|
||||||
|
if [[ "${1:-}" == "--force" ]]; then
|
||||||
|
FORCE=1
|
||||||
|
fi
|
||||||
|
|
||||||
command -v docker >/dev/null 2>&1 || {
|
command -v docker >/dev/null 2>&1 || {
|
||||||
printf '[%s] docker CLI unavailable; nothing to prune\n' "$(now)" >>"$LOG_FILE"
|
printf '[%s] docker CLI unavailable; nothing to prune\n' "$(now)" >>"$LOG_FILE"
|
||||||
exit 0
|
exit 0
|
||||||
}
|
}
|
||||||
|
|
||||||
printf '\n[%s] === docker prune start ===\n' "$(now)" >>"$LOG_FILE"
|
printf '\n[%s] === docker prune start%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
|
||||||
docker system df >>"$LOG_FILE" 2>&1 || true
|
docker system df >>"$LOG_FILE" 2>&1 || true
|
||||||
|
|
||||||
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
|
if (( FORCE )); then
|
||||||
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
|
docker builder prune -af >>"$LOG_FILE" 2>&1 || true
|
||||||
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
|
docker image prune -af >>"$LOG_FILE" 2>&1 || true
|
||||||
|
docker container prune -f >>"$LOG_FILE" 2>&1 || true
|
||||||
|
else
|
||||||
|
docker builder prune -af --filter "until=72h" --max-used-space=4g >>"$LOG_FILE" 2>&1 || true
|
||||||
|
docker image prune -af --filter "until=168h" >>"$LOG_FILE" 2>&1 || true
|
||||||
|
docker container prune -f --filter "until=24h" >>"$LOG_FILE" 2>&1 || true
|
||||||
|
fi
|
||||||
|
|
||||||
printf '\n[%s] === docker prune complete ===\n' "$(now)" >>"$LOG_FILE"
|
printf '\n[%s] === docker prune complete%s ===\n' "$(now)" "$( (( FORCE )) && printf ' (FORCE)' )" >>"$LOG_FILE"
|
||||||
docker system df >>"$LOG_FILE" 2>&1 || true
|
docker system df >>"$LOG_FILE" 2>&1 || true
|
||||||
+36
-4
@@ -17,7 +17,11 @@ import {
|
|||||||
healthDegraded,
|
healthDegraded,
|
||||||
} from "../src/lib/services/alert";
|
} from "../src/lib/services/alert";
|
||||||
import { runCatalogExport } from "../src/lib/services/catalog-git-export";
|
import { runCatalogExport } from "../src/lib/services/catalog-git-export";
|
||||||
import { diskLevel, parseDfOutput } from "../src/lib/services/disk-usage";
|
import {
|
||||||
|
DISK_THRESHOLDS,
|
||||||
|
diskLevel,
|
||||||
|
parseDfOutput,
|
||||||
|
} from "../src/lib/services/disk-usage";
|
||||||
import { invalidateNewsCache } from "../src/lib/services/news-cache";
|
import { invalidateNewsCache } from "../src/lib/services/news-cache";
|
||||||
import { rcon } from "../src/lib/services/rcon";
|
import { rcon } from "../src/lib/services/rcon";
|
||||||
|
|
||||||
@@ -133,6 +137,30 @@ async function checkDiskUsage(): Promise<void> {
|
|||||||
});
|
});
|
||||||
await diskPressure(usage);
|
await diskPressure(usage);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Self-healing: never let a mount max out. Safe prune from the warning
|
||||||
|
// mark, forced prune (drop age windows) from the error mark. Cooldown is
|
||||||
|
// keyed separately so a stuck fill level does not re-prune every run
|
||||||
|
// while still escalating to the forced path on real pressure.
|
||||||
|
if (usage.percent >= DISK_THRESHOLDS.error) {
|
||||||
|
if (canAlert(`disk-prune-force-${usage.mount}`)) {
|
||||||
|
logger.warn("Disk near full — forcing Docker cache reclaim", {
|
||||||
|
module: "jobs",
|
||||||
|
mount: usage.mount,
|
||||||
|
percent: usage.percent,
|
||||||
|
});
|
||||||
|
await pruneDockerCache(true);
|
||||||
|
}
|
||||||
|
} else if (usage.percent >= DISK_THRESHOLDS.warning) {
|
||||||
|
if (canAlert(`disk-prune-${usage.mount}`)) {
|
||||||
|
logger.info("Disk at high water mark — running safe Docker prune", {
|
||||||
|
module: "jobs",
|
||||||
|
mount: usage.mount,
|
||||||
|
percent: usage.percent,
|
||||||
|
});
|
||||||
|
await pruneDockerCache(false);
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -282,8 +310,10 @@ async function cleanupOldSessions(): Promise<void> {
|
|||||||
|
|
||||||
/** Host-side: reclaim Docker's unused cache (build cache, unreferenced images,
|
/** Host-side: reclaim Docker's unused cache (build cache, unreferenced images,
|
||||||
* stopped containers). Volumes and in-use images are never touched. No-op when
|
* stopped containers). Volumes and in-use images are never touched. No-op when
|
||||||
* docker or the prune script is unavailable. */
|
* docker or the prune script is unavailable. With force=true the age windows
|
||||||
async function pruneDockerCache(): Promise<void> {
|
* are dropped (docker-prune.sh --force) so every unused byte is reclaimed —
|
||||||
|
* the emergency path for a nearly-full disk. */
|
||||||
|
async function pruneDockerCache(force = false): Promise<void> {
|
||||||
const { access } = await import("node:fs/promises");
|
const { access } = await import("node:fs/promises");
|
||||||
const { resolve } = await import("node:path");
|
const { resolve } = await import("node:path");
|
||||||
const { spawn } = await import("node:child_process");
|
const { spawn } = await import("node:child_process");
|
||||||
@@ -294,7 +324,9 @@ async function pruneDockerCache(): Promise<void> {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
await new Promise<void>((resolvePromise) => {
|
await new Promise<void>((resolvePromise) => {
|
||||||
const child = spawn("bash", [script], { stdio: "ignore" });
|
const child = spawn("bash", [script, ...(force ? ["--force"] : [])], {
|
||||||
|
stdio: "ignore",
|
||||||
|
});
|
||||||
child.on("error", (err) =>
|
child.on("error", (err) =>
|
||||||
captureWorkerError(
|
captureWorkerError(
|
||||||
err,
|
err,
|
||||||
|
|||||||
@@ -36,6 +36,12 @@ it("preserves production runtime configuration and recent cache", () => {
|
|||||||
);
|
);
|
||||||
expect(prune).toContain('docker image prune -af --filter "until=168h"');
|
expect(prune).toContain('docker image prune -af --filter "until=168h"');
|
||||||
expect(prune).toContain('docker container prune -f --filter "until=24h"');
|
expect(prune).toContain('docker container prune -f --filter "until=24h"');
|
||||||
|
// Emergency `--force` mode drops every age window to reclaim unused bytes,
|
||||||
|
// but even then volumes are off-limits.
|
||||||
|
expect(prune).toContain('== "--force" ]]');
|
||||||
|
expect(prune).toContain("FORCE=1");
|
||||||
|
expect(prune).toContain("(( FORCE ))");
|
||||||
|
expect(deploy).not.toContain("--force");
|
||||||
expect(deploy).not.toContain("docker volume prune");
|
expect(deploy).not.toContain("docker volume prune");
|
||||||
expect(prune).not.toContain("docker volume prune");
|
expect(prune).not.toContain("docker volume prune");
|
||||||
});
|
});
|
||||||
|
|||||||
@@ -306,7 +306,7 @@ export function diskPressure(usage: {
|
|||||||
used: formatBytes(usage.usedBytes),
|
used: formatBytes(usage.usedBytes),
|
||||||
total: formatBytes(usage.totalBytes),
|
total: formatBytes(usage.totalBytes),
|
||||||
available: formatBytes(usage.availableBytes),
|
available: formatBytes(usage.availableBytes),
|
||||||
hint: "Docker prune runs nightly; run scripts/docker-prune.sh to reclaim cache sooner.",
|
hint: "Docker cache reclaim auto-fired with this alert; if the disk is still filling, the growth is outside Docker (check Gamedata/uploads/storage).",
|
||||||
},
|
},
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
Reference in new issue
Block a user