feat(ops): alert on filesystem fill levels from the host worker
Add a pure df parser (disk-usage.ts) with 85/90/95% threshold classification, a diskPressure() alert (Discord/email/alert_logs, severity escalates with fill), and a 5-minute host-side probe in jobs-worker.ts that raises one alert per crossing mount, cooldown-gated per mount+level. Real mounts only: overlay/tmpfs pseudo filesystems are ignored.
This commit is contained in:
1 parent
a0d7f42227
commit
fe26ca3ff3
5 files changed
+333
-2
No files matched your search
+55
-1
@@ -11,8 +11,13 @@ import {
|
||||
} from "../src/lib/db";
|
||||
import { logger } from "../src/lib/logger";
|
||||
import { redis } from "../src/lib/redis";
|
||||
import { emulatorOffline, healthDegraded } from "../src/lib/services/alert";
|
||||
import {
|
||||
diskPressure,
|
||||
emulatorOffline,
|
||||
healthDegraded,
|
||||
} from "../src/lib/services/alert";
|
||||
import { runCatalogExport } from "../src/lib/services/catalog-git-export";
|
||||
import { diskLevel, parseDfOutput } from "../src/lib/services/disk-usage";
|
||||
import { invalidateNewsCache } from "../src/lib/services/news-cache";
|
||||
import { rcon } from "../src/lib/services/rcon";
|
||||
|
||||
@@ -88,6 +93,49 @@ async function checkOpsHealth(): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Host-side: probe filesystem fill levels via `df -P -B1` and raise a
|
||||
* diskPressure() alert per mount once it crosses 85/90/95% (cooldown-gated per
|
||||
* mount+level so a stuck fill level does not spam Discord/email).
|
||||
*/
|
||||
async function checkDiskUsage(): Promise<void> {
|
||||
const { execFile } = await import("node:child_process");
|
||||
|
||||
let dfOut: string;
|
||||
try {
|
||||
dfOut = await new Promise<string>((resolve, reject) => {
|
||||
execFile(
|
||||
"df",
|
||||
["-P", "-B1"],
|
||||
{
|
||||
maxBuffer: 4 * 1024 * 1024,
|
||||
timeout: 15_000,
|
||||
},
|
||||
(err, stdout) => (err ? reject(err) : resolve(stdout)),
|
||||
);
|
||||
});
|
||||
} catch (err) {
|
||||
captureWorkerError(err, "Disk probe failed (is df available?)");
|
||||
return;
|
||||
}
|
||||
|
||||
const mounts = parseDfOutput(dfOut);
|
||||
if (mounts.length === 0) return;
|
||||
|
||||
for (const usage of mounts) {
|
||||
const level = diskLevel(usage.percent);
|
||||
if (level === 0) continue;
|
||||
if (canAlert(`disk-${usage.mount}-${level}`)) {
|
||||
logger.warn("Raising disk usage alert", {
|
||||
module: "jobs",
|
||||
mount: usage.mount,
|
||||
percent: usage.percent,
|
||||
});
|
||||
await diskPressure(usage);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
async function backupEmulatorJar(): Promise<void> {
|
||||
if (!env.EMULATOR_JAR_PATH || !env.EMULATOR_BACKUP_DIR) return;
|
||||
|
||||
@@ -351,6 +399,11 @@ async function main() {
|
||||
});
|
||||
logger.info("Scheduled: ops health probe (every 5 min)", { module: "jobs" });
|
||||
|
||||
new Cron("*/5 * * * *", () => {
|
||||
checkDiskUsage().catch((e) => captureWorkerError(e, "Disk check error"));
|
||||
});
|
||||
logger.info("Scheduled: disk usage probe (every 5 min)", { module: "jobs" });
|
||||
|
||||
new Cron("* * * * *", () => {
|
||||
publishScheduledArticles().catch((e) =>
|
||||
captureWorkerError(e, "ScheduledArticlePublish"),
|
||||
@@ -369,6 +422,7 @@ async function main() {
|
||||
cleanupOldLogs(),
|
||||
cleanupOldSessions(),
|
||||
checkOpsHealth(),
|
||||
checkDiskUsage(),
|
||||
]);
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user